Commit 08a7e391b14 for nodejs

commit 08a7e391b14f0889e0a3b5c4a3f02633070acb8b
Author: Daniel Lemire <daniel@lemire.me>
Date:   Thu Oct 8 20:01:38 2026 -0400

    deps: update simdjson to 5.0.3

    PR-URL: https://github.com/nodejs/node/pull/66620
    Refs: https://github.com/simdjson/simdjson/pull/2913
    Refs: https://github.com/nodejs/node/pull/66495
    Reviewed-By: Yagiz Nizipli <yagiz@nizipli.com>
    Reviewed-By: Antoine du Hamel <duhamelantoine1995@gmail.com>
    Reviewed-By: Richard Lau <richard.lau@ibm.com>

diff --git a/deps/simdjson/simdjson.cpp b/deps/simdjson/simdjson.cpp
index 71f443b3d2b..980bb83cefa 100644
--- a/deps/simdjson/simdjson.cpp
+++ b/deps/simdjson/simdjson.cpp
@@ -1,4 +1,4 @@
-/* auto-generated on 2026-09-04 16:04:31 -0400. version 4.6.11 Do not edit! */
+/* auto-generated on 2026-10-07 22:43:29 -0400. version 5.0.3 Do not edit! */
 /* including simdjson.cpp:  */
 /* begin file simdjson.cpp */
 #define SIMDJSON_SRC_SIMDJSON_CPP
@@ -41,7 +41,9 @@
 #endif

 // C++ 26
-#if !defined(SIMDJSON_CPLUSPLUS26) && (SIMDJSON_CPLUSPLUS >= 202402L) // update when the standard is finalized
+// While C++26 is a working draft, compilers report 202400L in C++26 mode
+// (both GCC 16 and Clang 21 do). Update when the standard is finalized.
+#if !defined(SIMDJSON_CPLUSPLUS26) && (SIMDJSON_CPLUSPLUS >= 202400L)
 #define SIMDJSON_CPLUSPLUS26 1
 #endif

@@ -98,14 +100,48 @@
 #endif
 #endif

-// The current specification is unclear on how we detect
-// static reflection, both __cpp_lib_reflection and
-// __cpp_impl_reflection are proposed in the draft specification.
-// For now, we disable static reflect by default. It must be
-// specified at compiler time.
+// Static reflection.
+//
+// The reflection-based APIs (simdjson::to, document::get<T>, the builder,
+// compile-time JSON, annotations) need considerably more than the reflection
+// operator. We turn them on only when the compiler advertises all of:
+//
+//   P2996 reflection (^^, splicers, <meta>)  __cpp_impl_reflection,
+//                                            __cpp_lib_reflection
+//   P1306 expansion statements (template for) __cpp_expansion_statements
+//   P3491 std::define_static_string / _array  __cpp_lib_define_static
+//
+// Two further features we rely on have, as of this writing, no feature-test
+// macro of their own, so they cannot be checked directly:
+//
+//   P3394 annotations ([[=x]], std::meta::annotations_of) -- used for
+//         the annotations of simdjson/annotations.h (rename, skip, ...).
+//   P3289 consteval blocks (consteval { ... }) -- used by compile_time_json.
+//
+// Every implementation that defines the four macros above also implements
+// those two, so requiring the four is sufficient in practice. If that ever
+// stops being true, define SIMDJSON_STATIC_REFLECTION=0 to opt out.
+//
+// SIMDJSON_STATIC_REFLECTION may always be defined by the user (or by the
+// build system) to 0 or 1 to override the detection.
+//
+// Note that C++26 mode alone is not enough: GCC 16 requires -freflection,
+// and only then does it define __cpp_impl_reflection.
 #ifndef SIMDJSON_STATIC_REFLECTION
-#define SIMDJSON_STATIC_REFLECTION 0 // disabled by default.
+#if defined(SIMDJSON_CPLUSPLUS26) &&                                           \
+    defined(__cpp_impl_reflection) && __cpp_impl_reflection >= 202506L &&      \
+    defined(__cpp_lib_reflection) && __cpp_lib_reflection >= 202506L &&        \
+    defined(__cpp_expansion_statements) &&                                     \
+        __cpp_expansion_statements >= 202506L &&                               \
+    defined(__cpp_lib_define_static) && __cpp_lib_define_static >= 202506L
+// __cpp_lib_reflection is the feature-test macro for <meta>, so there is no
+// need for a separate __has_include check (which would have to be guarded for
+// compilers that lack __has_include).
+#define SIMDJSON_STATIC_REFLECTION 1
+#else
+#define SIMDJSON_STATIC_REFLECTION 0
 #endif
+#endif // SIMDJSON_STATIC_REFLECTION

 #if defined(__apple_build_version__)
 #if __apple_build_version__ < 14000000
@@ -138,6 +174,47 @@
 #define SIMDJSON_SUPPORTS_DESERIALIZATION 0
 #endif

+// The C++20 char8_t type (and std::u8string/std::u8string_view) is available.
+// Because all strings that simdjson produces are valid UTF-8, we can offer
+// char8_t variants of our string accessors when this macro is set.
+#if !defined(SIMDJSON_SUPPORTS_CHAR8_T)
+#if defined(__cpp_char8_t) && __cpp_char8_t >= 201811L
+#define SIMDJSON_SUPPORTS_CHAR8_T 1
+#else
+#define SIMDJSON_SUPPORTS_CHAR8_T 0
+#endif
+#endif // !defined(SIMDJSON_SUPPORTS_CHAR8_T)
+
+// The C++23 fixed-width floating-point types std::float32_t and std::float64_t
+// (<stdfloat>) are available. They are optional even in C++23: a compiler that
+// provides them predefines __STDCPP_FLOAT32_T__ and __STDCPP_FLOAT64_T__.
+// When these macros are set, we offer get_float32() and get_float64().
+#if !defined(SIMDJSON_SUPPORTS_FLOAT32_T)
+#if defined(__STDCPP_FLOAT32_T__) && defined(__has_include)
+#if __has_include(<stdfloat>)
+#define SIMDJSON_SUPPORTS_FLOAT32_T 1
+#endif
+#endif
+#ifndef SIMDJSON_SUPPORTS_FLOAT32_T
+#define SIMDJSON_SUPPORTS_FLOAT32_T 0
+#endif
+#endif // !defined(SIMDJSON_SUPPORTS_FLOAT32_T)
+
+#if !defined(SIMDJSON_SUPPORTS_FLOAT64_T)
+#if defined(__STDCPP_FLOAT64_T__) && defined(__has_include)
+#if __has_include(<stdfloat>)
+#define SIMDJSON_SUPPORTS_FLOAT64_T 1
+#endif
+#endif
+#ifndef SIMDJSON_SUPPORTS_FLOAT64_T
+#define SIMDJSON_SUPPORTS_FLOAT64_T 0
+#endif
+#endif // !defined(SIMDJSON_SUPPORTS_FLOAT64_T)
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T || SIMDJSON_SUPPORTS_FLOAT64_T
+#include <stdfloat>
+#endif
+

 #if !defined(SIMDJSON_CONSTEVAL)
 #if defined(__cpp_consteval) && __cpp_consteval >= 201811L && defined(__cpp_lib_constexpr_string) && __cpp_lib_constexpr_string >= 201907L
@@ -146,6 +223,18 @@
 #define SIMDJSON_CONSTEVAL 0
 #endif // defined(__cpp_consteval) && __cpp_consteval >= 201811L && defined(__cpp_lib_constexpr_string) && __cpp_lib_constexpr_string >= 201907L
 #endif // !defined(SIMDJSON_CONSTEVAL)
+
+// SIMDJSON_CONSTEXPR_STRING is 'constexpr' when the standard library supports
+// constexpr std::string (e.g., libstdc++ 12 or better), and empty otherwise. It
+// lets functions that build a std::string be constant expressions when possible
+// while still compiling against older standard libraries.
+#if !defined(SIMDJSON_CONSTEXPR_STRING)
+#if defined(__cpp_lib_constexpr_string) && __cpp_lib_constexpr_string >= 201907L
+#define SIMDJSON_CONSTEXPR_STRING constexpr
+#else
+#define SIMDJSON_CONSTEXPR_STRING
+#endif // defined(__cpp_lib_constexpr_string) && __cpp_lib_constexpr_string >= 201907L
+#endif // !defined(SIMDJSON_CONSTEXPR_STRING)
 #endif // SIMDJSON_COMPILER_CHECK_H
 /* end file simdjson/compiler_check.h */
 /* including simdjson/portability.h: #include "simdjson/portability.h" */
@@ -437,16 +526,86 @@ using std::size_t;
 #endif
 #endif

+#ifndef SIMDJSON_HAS_UNISTD_H
+#if defined(__unix__) || defined(__APPLE__) || defined(__linux__)
+#define SIMDJSON_HAS_UNISTD_H 1
+#else
+#define SIMDJSON_HAS_UNISTD_H 0
+#endif
+#endif
+
+// padded_memory_map availability.
+//
+// On POSIX platforms the class is always available: the implementation uses
+// `mmap` (and a trailing anonymous page for padding) from <sys/mman.h>.
+//
+// On Windows the class is disabled by default and must be explicitly
+// opted into by defining `SIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS=1`. Enabling
+// it requires:
+//   1. `<windows.h>` has been included *before* `<simdjson.h>` (so that
+//      this header can see the Win32 types and the `_WINDOWS_` include
+//      guard),
+//   2. the compilation targets Windows 10, version 1803 or later
+//      (i.e. `NTDDI_VERSION >= NTDDI_WIN10_RS4`, `0x0A000005`). This is
+//      required because the implementation relies on the modern memory
+//      APIs introduced with that version (`CreateFileMapping2` /
+//      `MapViewOfFile3`),
+//   3. the link step pulls in an import library that exports those APIs,
+//      typically `onecore.lib` (or `mincore.lib`).
+//
+// The `SIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS` CMake option arranges (1)-(3)
+// automatically when building simdjson with its own CMake. Consumers using
+// simdjson as a pre-built library are responsible for setting the macro,
+// the Windows version macros, and the link library themselves.
+//
+// If the opt-in conditions are not met on Windows, `padded_memory_map`
+// simply does not exist -- any attempt to use it fails at compile time
+// with an "unknown identifier" diagnostic rather than silently degrading.
+//
+// The SIMDJSON_HAS_PADDED_MEMORY_MAP macro reflects whether the class is
+// available in the current translation unit. Users may test this macro to
+// conditionally compile code that depends on padded_memory_map.
+#ifndef SIMDJSON_HAS_PADDED_MEMORY_MAP
+  #if defined(__unix__) || defined(__APPLE__) || defined(__linux__)
+    #define SIMDJSON_HAS_PADDED_MEMORY_MAP 1
+  #elif defined(_WINDOWS_) && defined(SIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS) && SIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS
+    #define SIMDJSON_HAS_PADDED_MEMORY_MAP 1
+  #else
+    #define SIMDJSON_HAS_PADDED_MEMORY_MAP 0
+  #endif
+#endif

 #endif // SIMDJSON_PORTABILITY_H
 /* end file simdjson/portability.h */
+#include <cstddef>

 namespace simdjson {
 namespace internal {
+/**
+ * @private
+ * Scratch capacity that every caller of to_chars must provide.
+ *
+ * The emitted decimal is at most ~24 characters, but dragonbox() and
+ * format_buffer() intentionally write past the logical end with fixed-size
+ * 16/17-byte memcpy/memset operations so the compiler can inline them (no
+ * libc mem* dispatch with size-class branches). The extra bytes are required
+ * for safety of those over-writes; do not shrink this below 40.
+ * See src/to_chars.cpp and #2805.
+ */
+// Use an unscoped enum (not static constexpr / inline constexpr):
+// - C++11 targets (readme_examples11, quickstart11, ...) still include this header
+// - a static constexpr in the amalgamated simdjson.cpp TU is unused there
+//   (only callers in headers use it) and trips -Wunused-const-variable -Werror
+enum : size_t { to_chars_buffer_size = 40 };
 /**
  * @private
  * Our own implementation of the C++17 to_chars function.
  * Defined in src/to_chars
+ *
+ * @note The buffer starting at first must have at least to_chars_buffer_size
+ *       bytes of writable storage (see to_chars_buffer_size).
+ * @note The input number must be finite (NaN/Inf are not supported).
+ * @note The result is NOT null-terminated.
  */
 char *to_chars(char *first, const char *last, double value);
 /**
@@ -456,6 +615,12 @@ char *to_chars(char *first, const char *last, double value);
  */
 double from_chars(const char *first) noexcept;
 double from_chars(const char *first, const char* end) noexcept;
+/**
+ * @private
+ * Same as from_chars, but produces a correctly rounded binary32 (float) value.
+ * Defined in src/from_chars
+ */
+float from_chars_float(const char *first) noexcept;
 }

 #ifndef SIMDJSON_EXCEPTIONS
@@ -466,6 +631,10 @@ double from_chars(const char *first, const char* end) noexcept;
 #endif
 #endif

+#ifndef SIMDJSON_ENABLE_NAN_INF
+#define SIMDJSON_ENABLE_NAN_INF 0
+#endif
+
 } // namespace simdjson

 #if defined(__GNUC__)
@@ -481,16 +650,14 @@ double from_chars(const char *first, const char* end) noexcept;

 // Align to N-byte boundary
 #define SIMDJSON_ROUNDUP_N(a, n) (((a) + ((n)-1)) & ~((n)-1))
-#define SIMDJSON_ROUNDDOWN_N(a, n) ((a) & ~((n)-1))
-
-#define SIMDJSON_ISALIGNED_N(ptr, n) (((uintptr_t)(ptr) & ((n)-1)) == 0)

 #if SIMDJSON_REGULAR_VISUAL_STUDIO
   // We could use [[deprecated]] but it requires C++14
   #define simdjson_deprecated __declspec(deprecated)

   #define simdjson_really_inline __forceinline
-  #define simdjson_never_inline __declspec(noinline)
+  #define simdjson_never_inline inline __declspec(noinline)
+  #define simdjson_really_flatten [[msvc::flatten]]

   #define simdjson_unused
   #define simdjson_warn_unused
@@ -531,6 +698,7 @@ double from_chars(const char *first, const char* end) noexcept;

   #define simdjson_really_inline inline __attribute__((always_inline))
   #define simdjson_never_inline inline __attribute__((noinline))
+  #define simdjson_really_flatten [[gnu::flatten]]

   #define simdjson_unused __attribute__((unused))
   #define simdjson_warn_unused __attribute__((warn_unused_result))
@@ -607,6 +775,15 @@ double from_chars(const char *first, const char* end) noexcept;
   #define simdjson_inline simdjson_really_inline
 #endif

+#if defined(simdjson_flatten)
+  // Prefer the user's definition of simdjson_flatten; don't define it ourselves.
+#elif (defined(__GNUC__) && !defined(__OPTIMIZE__)) || (defined(_DEBUG) && _MSC_VER )
+  // Flattening can lead to significant code bloat and high compile times. Don't use it for unoptimized builds.
+  #define simdjson_flatten
+#else
+  #define simdjson_flatten simdjson_really_flatten
+#endif
+
 #if SIMDJSON_VISUAL_STUDIO
     /**
      * Windows users need to do some extra work when building
@@ -2562,6 +2739,7 @@ enum error_code {
   OUT_OF_BOUNDS,              ///< Attempted to access location outside of document.
   TRAILING_CONTENT,           ///< Unexpected trailing content in the JSON input
   OUT_OF_CAPACITY,            ///< The capacity was exceeded, we cannot allocate enough memory.
+  UNKNOWN_FIELD,              ///< JSON field does not map to any member of the target (see simdjson::deny_unknown_fields)
   NUM_ERROR_CODES             ///< Placeholder for end of error code list.
 };

@@ -2924,6 +3102,7 @@ inline const std::string error_message(int error) noexcept;
 #if SIMDJSON_SUPPORTS_CONCEPTS

 #include <concepts>
+#include <string_view>
 #include <type_traits>

 namespace simdjson {
@@ -2965,6 +3144,19 @@ concept constructible_from_string_view = std::is_constructible_v<T, std::string_
                                         && !std::is_same_v<T, std::string_view>
                                         && std::is_default_constructible_v<T>;

+#if SIMDJSON_SUPPORTS_CHAR8_T
+/**
+ * A C++20 char8_t string type such as std::u8string. Such types cannot be built
+ * from a std::string_view (the character types differ), so they need their own
+ * deserialization path, going through the u8 string accessors.
+ */
+template<typename T>
+concept constructible_from_u8string_view = std::is_constructible_v<T, std::u8string_view>
+                                        && !std::is_same_v<T, std::u8string_view>
+                                        && !std::is_constructible_v<T, std::string_view>
+                                        && std::is_default_constructible_v<T>;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 template<typename M>
 concept string_view_keyed_map = string_view_like<typename M::key_type>
               && requires(std::remove_cvref_t<M>& m, typename M::key_type sv, typename M::mapped_type v) {
@@ -3059,9 +3251,15 @@ concept string_like =
 // Concept that checks if a type is a container but not a string (because
 // strings handling must be handled differently)
 // Now uses iterator-based approach for broader container support
+//
+// Optional types are excluded on purpose. Since C++26 (P3168), std::optional
+// is itself a range, so without the exclusion an std::optional would match
+// both this concept and optional_type, making the container and the optional
+// overloads of atom()/append() ambiguous. See issue 2827.
 template <typename T>
 concept container_but_not_string =
-  std::ranges::input_range<T> && !string_like<T> && !concepts::string_view_keyed_map<T>;
+  std::ranges::input_range<T> && !string_like<T> && !concepts::string_view_keyed_map<T>
+  && !concepts::optional_type<T>;



@@ -3168,6 +3366,11 @@ struct fixed_string {
             data[i] = str[i];
         }
     }
+    constexpr fixed_string(const unsigned char (&str)[N])  {
+        for (std::size_t i = 0; i < N; ++i) {
+            data[i] = static_cast<char>(str[i]);
+        }
+    }
     char data[N];
     constexpr std::string_view view() const { return {data, N - 1}; }
     constexpr size_t size() const { return N ; }
@@ -3199,6 +3402,11 @@ struct string_constant {
 #endif // SIMDJSON_CONSTEVALUTIL_H
 /* end file simdjson/constevalutil.h */

+#if SIMDJSON_SUPPORTS_CHAR8_T
+#include <string>
+#include <string_view>
+#endif
+
 /**
  * @brief The top level simdjson namespace, containing everything the library provides.
  */
@@ -3208,6 +3416,10 @@ SIMDJSON_PUSH_DISABLE_UNUSED_WARNINGS

 /** The maximum document size supported by simdjson. */
 constexpr size_t SIMDJSON_MAXSIZE_BYTES = 0xFFFFFFFF;
+/** The maximum depth of nested objects and arrays supported by simdjson.
+ A depth of SIMDJSON_MAXSIZE_BYTES/2 is not reasonable and would be
+ adversarial, but it serves as an upper bound for validation purposes. */
+constexpr size_t SIMDJSON_MAX_DEPTH = SIMDJSON_MAXSIZE_BYTES/2;

 /**
  * The amount of padding needed in a buffer to parse JSON.
@@ -3233,6 +3445,30 @@ struct padded_string;
 class padded_string_view;
 enum class stage1_mode;

+/**
+ * Stream format for parse_many/iterate_many.
+ */
+enum class stream_format {
+  whitespace_delimited, ///< Whitespace-delimited JSON documents (default, includes NDJSON/JSONL)
+  json_sequence,        ///< RFC 7464 JSON text sequences (RS-delimited)
+  comma_delimited,      ///< Comma-separated JSON documents (e.g., `{...},{...},{...}`)
+  comma_delimited_array,///< A single JSON array whose elements are iterated as
+                        ///< comma-separated documents (e.g., `[{...},{...},{...}]`).
+                        ///< The parser strips the outer `[` / `]` plus any
+                        ///< surrounding JSON whitespace (space, tab, LF, CR)
+                        ///< and then behaves like `comma_delimited` over the
+                        ///< remaining bytes.
+  newline_delimited     ///< NDJSON/JSON Lines where each document occupies exactly
+                        ///< one line: documents are separated by line feeds and no
+                        ///< document contains a raw line feed. Same inputs as
+                        ///< `whitespace_delimited`, but the stronger guarantee lets
+                        ///< the parser find the end of a document without walking
+                        ///< it. On ondemand `iterate_many`, an unread remainder may
+                        ///< be skipped by jumping to the next line feed without
+                        ///< structure-validating that remainder. Use
+                        ///< `whitespace_delimited` if unsure.
+};
+
 namespace internal {

 template<typename T>
@@ -3243,6 +3479,52 @@ class tape_ref;
 struct value128;
 enum class tape_type;

+#if SIMDJSON_SUPPORTS_CHAR8_T
+/**
+ * Reinterpret a UTF-8 string as a C++20 std::u8string_view. No byte is copied
+ * or modified: char8_t and char have the same size, representation and
+ * alignment. Every string that simdjson produces is valid UTF-8, so this is a
+ * lossless view over the very same memory.
+ * @private
+ */
+simdjson_inline std::u8string_view as_u8string_view(std::string_view v) noexcept {
+  return std::u8string_view(reinterpret_cast<const char8_t *>(v.data()), v.size());
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
+/**
+ * Assign a UTF-8 string to a string-like receiver. The general case simply
+ * assigns the std::string_view: it covers std::string and any user type that
+ * can be assigned from a std::string_view.
+ * @private
+ */
+template <typename string_type>
+simdjson_inline void assign_utf8(string_type &receiver, std::string_view content) noexcept {
+  receiver = content;
+}
+
+#if SIMDJSON_SUPPORTS_CHAR8_T
+/**
+ * Assign a UTF-8 string to a char8_t-based string (e.g., std::u8string). This
+ * overload is more specialized than the general one, so overload resolution
+ * prefers it whenever the receiver holds char8_t.
+ * @private
+ */
+template <typename traits_type, typename allocator_type>
+simdjson_inline void assign_utf8(std::basic_string<char8_t, traits_type, allocator_type> &receiver, std::string_view content) noexcept {
+  receiver.assign(reinterpret_cast<const char8_t *>(content.data()), content.size());
+}
+
+/**
+ * Assign a UTF-8 string to a char8_t-based string view (e.g., std::u8string_view).
+ * @private
+ */
+template <typename traits_type>
+simdjson_inline void assign_utf8(std::basic_string_view<char8_t, traits_type> &receiver, std::string_view content) noexcept {
+  receiver = std::basic_string_view<char8_t, traits_type>(reinterpret_cast<const char8_t *>(content.data()), content.size());
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 } // namespace internal
 } // namespace simdjson

@@ -3263,806 +3545,1038 @@ SIMDJSON_PUSH_DISABLE_UNUSED_WARNINGS

 #include <cstring>
 #include <cstdint>
-#include <array>
 #include <cmath>
+#include <limits>

 namespace simdjson {
 namespace internal {
 /*!
-implements the Grisu2 algorithm for binary to decimal floating-point
-conversion.
-Adapted from JSON for Modern C++
-
-This implementation is a slightly modified version of the reference
-implementation which may be obtained from
-http://florian.loitsch.com/publications (bench.tar.gz).
-The code is distributed under the MIT license, Copyright (c) 2009 Florian
-Loitsch. For a detailed description of the algorithm see: [1] Loitsch, "Printing
-Floating-Point Numbers Quickly and Accurately with Integers", Proceedings of the
-ACM SIGPLAN 2010 Conference on Programming Language Design and Implementation,
-PLDI 2010 [2] Burger, Dybvig, "Printing Floating-Point Numbers Quickly and
-Accurately", Proceedings of the ACM SIGPLAN 1996 Conference on Programming
-Language Design and Implementation, PLDI 1996
-*/
-namespace dtoa_impl {
-
-template <typename Target, typename Source>
-Target reinterpret_bits(const Source source) {
-  static_assert(sizeof(Target) == sizeof(Source), "size mismatch");
-
-  Target target;
-  std::memcpy(&target, &source, sizeof(Source));
-  return target;
-}
-
-struct diyfp // f * 2^e
-{
-  static constexpr int kPrecision = 64; // = q
-
-  std::uint64_t f = 0;
-  int e = 0;
+Implements the Dragonbox algorithm for binary to decimal floating-point
+conversion (shortest round-trip representation of a double).

-  constexpr diyfp(std::uint64_t f_, int e_) noexcept : f(f_), e(e_) {}
+The digit-generation core below is a self-contained port of Junekey Jeon's
+reference "simple_dragonbox" implementation, specialized to IEEE-754 binary64
+and de-templated to match simdjson's style. Only the shortest-representation
+path with the default (nearest, ties-to-even) rounding is kept.

-  /*!
-  @brief returns x - y
-  @pre x.e == y.e and x.f >= y.f
-  */
-  static diyfp sub(const diyfp &x, const diyfp &y) noexcept {
+Dragonbox: https://github.com/jk-jeon/dragonbox
+Copyright 2020-2025 Junekey Jeon (and contributors).

-    return {x.f - y.f, x.e};
-  }
+The original is dual-licensed; this port is used under the terms of the
+Boost Software License, Version 1.0 (https://www.boost.org/LICENSE_1_0.txt).

-  /*!
-  @brief returns x * y
-  @note The result is rounded. (Only the upper q bits are returned.)
-  */
-  static diyfp mul(const diyfp &x, const diyfp &y) noexcept {
-    static_assert(kPrecision == 64, "internal error");
+For the algorithm itself see:
+[1] Junekey Jeon, "Dragonbox: A New Floating-Point Binary-to-Decimal Conversion Algorithm" (2022).

-    // Computes:
-    //  f = round((x.f * y.f) / 2^q)
-    //  e = x.e + y.e + q
-
-    // Emulate the 64-bit * 64-bit multiplication:
-    //
-    // p = u * v
-    //   = (u_lo + 2^32 u_hi) (v_lo + 2^32 v_hi)
-    //   = (u_lo v_lo         ) + 2^32 ((u_lo v_hi         ) + (u_hi v_lo )) +
-    //   2^64 (u_hi v_hi         ) = (p0                ) + 2^32 ((p1 ) + (p2 ))
-    //   + 2^64 (p3                ) = (p0_lo + 2^32 p0_hi) + 2^32 ((p1_lo +
-    //   2^32 p1_hi) + (p2_lo + 2^32 p2_hi)) + 2^64 (p3                ) =
-    //   (p0_lo             ) + 2^32 (p0_hi + p1_lo + p2_lo ) + 2^64 (p1_hi +
-    //   p2_hi + p3) = (p0_lo             ) + 2^32 (Q ) + 2^64 (H ) = (p0_lo ) +
-    //   2^32 (Q_lo + 2^32 Q_hi                           ) + 2^64 (H )
-    //
-    // (Since Q might be larger than 2^32 - 1)
-    //
-    //   = (p0_lo + 2^32 Q_lo) + 2^64 (Q_hi + H)
-    //
-    // (Q_hi + H does not overflow a 64-bit int)
-    //
-    //   = p_lo + 2^64 p_hi
-
-    const std::uint64_t u_lo = x.f & 0xFFFFFFFFu;
-    const std::uint64_t u_hi = x.f >> 32u;
-    const std::uint64_t v_lo = y.f & 0xFFFFFFFFu;
-    const std::uint64_t v_hi = y.f >> 32u;
-
-    const std::uint64_t p0 = u_lo * v_lo;
-    const std::uint64_t p1 = u_lo * v_hi;
-    const std::uint64_t p2 = u_hi * v_lo;
-    const std::uint64_t p3 = u_hi * v_hi;
-
-    const std::uint64_t p0_hi = p0 >> 32u;
-    const std::uint64_t p1_lo = p1 & 0xFFFFFFFFu;
-    const std::uint64_t p1_hi = p1 >> 32u;
-    const std::uint64_t p2_lo = p2 & 0xFFFFFFFFu;
-    const std::uint64_t p2_hi = p2 >> 32u;
-
-    std::uint64_t Q = p0_hi + p1_lo + p2_lo;
-
-    // The full product might now be computed as
-    //
-    // p_hi = p3 + p2_hi + p1_hi + (Q >> 32)
-    // p_lo = p0_lo + (Q << 32)
-    //
-    // But in this particular case here, the full p_lo is not required.
-    // Effectively we only need to add the highest bit in p_lo to p_hi (and
-    // Q_hi + 1 does not overflow).
-
-    Q += std::uint64_t{1} << (64u - 32u - 1u); // round, ties up
-
-    const std::uint64_t h = p3 + p2_hi + p1_hi + (Q >> 32u);
-
-    return {h, x.e + y.e + 64};
-  }
-
-  /*!
-  @brief normalize x such that the significand is >= 2^(q-1)
-  @pre x.f != 0
-  */
-  static diyfp normalize(diyfp x) noexcept {
-
-    while ((x.f >> 63u) == 0) {
-      x.f <<= 1u;
-      x.e--;
-    }
-
-    return x;
-  }
-
-  /*!
-  @brief normalize x such that the result has the exponent E
-  @pre e >= x.e and the upper e - x.e bits of x.f must be zero.
-  */
-  static diyfp normalize_to(const diyfp &x,
-                            const int target_exponent) noexcept {
-    const int delta = x.e - target_exponent;
-
-    return {x.f << delta, target_exponent};
-  }
-};
-
-struct boundaries {
-  diyfp w;
-  diyfp minus;
-  diyfp plus;
-};
-
-/*!
-Compute the (normalized) diyfp representing the input number 'value' and its
-boundaries.
-@pre value must be finite and positive
+The shortest decimal digits produced here are then laid out into the familiar
+printf("%g")-style text by format_buffer(), which is unchanged from the previous
+Grisu2-based implementation, so the emitted strings are identical except that
+Dragonbox always yields the (sometimes shorter) shortest representation.
 */
-template <typename FloatType> boundaries compute_boundaries(FloatType value) {
+namespace dtoa_impl {

-  // Convert the IEEE representation into a diyfp.
-  //
-  // If v is denormal:
-  //      value = 0.F * 2^(1 - bias) = (          F) * 2^(1 - bias - (p-1))
-  // If v is normalized:
-  //      value = 1.F * 2^(E - bias) = (2^(p-1) + F) * 2^(E - bias - (p-1))
+// 128-bit helpers (no compiler intrinsics, so the code stays portable).
+struct uint128 {
+  std::uint64_t high;
+  std::uint64_t low;
+};

-  static_assert(std::numeric_limits<FloatType>::is_iec559,
-                "internal error: dtoa_short requires an IEEE-754 "
-                "floating-point implementation");
+inline std::uint64_t rotr64(std::uint64_t n, unsigned r) noexcept {
+  r &= 63;
+  return (n >> r) | (n << ((64 - r) & 63));
+}

-  constexpr int kPrecision =
-      std::numeric_limits<FloatType>::digits; // = p (includes the hidden bit)
-  constexpr int kBias =
-      std::numeric_limits<FloatType>::max_exponent - 1 + (kPrecision - 1);
-  constexpr int kMinExp = 1 - kBias;
-  constexpr std::uint64_t kHiddenBit = std::uint64_t{1}
-                                       << (kPrecision - 1); // = 2^(p-1)
+inline std::uint64_t umul64(std::uint32_t x, std::uint32_t y) noexcept {
+  return x * std::uint64_t(y);
+}

-  using bits_type = typename std::conditional<kPrecision == 24, std::uint32_t,
-                                              std::uint64_t>::type;
+// 64x64 -> 128 bit multiplication.
+inline uint128 umul128(std::uint64_t x, std::uint64_t y) noexcept {
+#if defined(__SIZEOF_INT128__)
+  const __uint128_t p = static_cast<__uint128_t>(x)*y;
+  return {std::uint64_t(p>>64), std::uint64_t(p)};
+#else // using fallback on 32-bit targets and MSVC
+  const std::uint32_t a = std::uint32_t(x >> 32);
+  const std::uint32_t b = std::uint32_t(x);
+  const std::uint32_t c = std::uint32_t(y >> 32);
+  const std::uint32_t d = std::uint32_t(y);

-  const std::uint64_t bits = reinterpret_bits<bits_type>(value);
-  const std::uint64_t E = bits >> (kPrecision - 1);
-  const std::uint64_t F = bits & (kHiddenBit - 1);
+  const std::uint64_t ac = umul64(a, c);
+  const std::uint64_t bc = umul64(b, c);
+  const std::uint64_t ad = umul64(a, d);
+  const std::uint64_t bd = umul64(b, d);

-  const bool is_denormal = E == 0;
-  const diyfp v = is_denormal
-                      ? diyfp(F, kMinExp)
-                      : diyfp(F + kHiddenBit, static_cast<int>(E) - kBias);
+  const std::uint64_t intermediate =
+      (bd >> 32) + std::uint32_t(ad) + std::uint32_t(bc);

-  // Compute the boundaries m- and m+ of the floating-point value
-  // v = f * 2^e.
-  //
-  // Determine v- and v+, the floating-point predecessor and successor if v,
-  // respectively.
-  //
-  //      v- = v - 2^e        if f != 2^(p-1) or e == e_min                (A)
-  //         = v - 2^(e-1)    if f == 2^(p-1) and e > e_min                (B)
-  //
-  //      v+ = v + 2^e
-  //
-  // Let m- = (v- + v) / 2 and m+ = (v + v+) / 2. All real numbers _strictly_
-  // between m- and m+ round to v, regardless of how the input rounding
-  // algorithm breaks ties.
-  //
-  //      ---+-------------+-------------+-------------+-------------+---  (A)
-  //         v-            m-            v             m+            v+
-  //
-  //      -----------------+------+------+-------------+-------------+---  (B)
-  //                       v-     m-     v             m+            v+
+  return {ac + (intermediate >> 32) + (ad >> 32) + (bc >> 32),
+          (intermediate << 32) + std::uint32_t(bd)};
+#endif
+}

-  const bool lower_boundary_is_closer = F == 0 && E > 1;
-  const diyfp m_plus = diyfp(2 * v.f + 1, v.e - 1);
-  const diyfp m_minus = lower_boundary_is_closer
-                            ? diyfp(4 * v.f - 1, v.e - 2)  // (B)
-                            : diyfp(2 * v.f - 1, v.e - 1); // (A)
+// High 64 bits of a 64x64 -> 128 bit multiplication.
+inline std::uint64_t umul128_upper64(std::uint64_t x, std::uint64_t y) noexcept {
+#if defined(__SIZEOF_INT128__)
+  return std::uint64_t((static_cast<__uint128_t>(x)*y)>>64);
+#else // using fallback on 32-bit targets and MSVC
+  const std::uint32_t a = std::uint32_t(x >> 32);
+  const std::uint32_t b = std::uint32_t(x);
+  const std::uint32_t c = std::uint32_t(y >> 32);
+  const std::uint32_t d = std::uint32_t(y);

-  // Determine the normalized w+ = m+.
-  const diyfp w_plus = diyfp::normalize(m_plus);
+  const std::uint64_t ac = umul64(a, c);
+  const std::uint64_t bc = umul64(b, c);
+  const std::uint64_t ad = umul64(a, d);
+  const std::uint64_t bd = umul64(b, d);

-  // Determine w- = m- such that e_(w-) = e_(w+).
-  const diyfp w_minus = diyfp::normalize_to(m_minus, w_plus.e);
+  const std::uint64_t intermediate =
+      (bd >> 32) + std::uint32_t(ad) + std::uint32_t(bc);

-  return {diyfp::normalize(v), w_minus, w_plus};
+  return ac + (intermediate >> 32) + (ad >> 32) + (bc >> 32);
+#endif
 }

-// Given normalized diyfp w, Grisu needs to find a (normalized) cached
-// power-of-ten c, such that the exponent of the product c * w = f * 2^e lies
-// within a certain range [alpha, gamma] (Definition 3.2 from [1])
-//
-//      alpha <= e = e_c + e_w + q <= gamma
-//
-// or
-//
-//      f_c * f_w * 2^alpha <= f_c 2^(e_c) * f_w 2^(e_w) * 2^q
-//                          <= f_c * f_w * 2^gamma
-//
-// Since c and w are normalized, i.e. 2^(q-1) <= f < 2^q, this implies
-//
-//      2^(q-1) * 2^(q-1) * 2^alpha <= c * w * 2^q < 2^q * 2^q * 2^gamma
-//
-// or
-//
-//      2^(q - 2 + alpha) <= c * w < 2^(q + gamma)
-//
-// The choice of (alpha,gamma) determines the size of the table and the form of
-// the digit generation procedure. Using (alpha,gamma)=(-60,-32) works out well
-// in practice:
-//
-// The idea is to cut the number c * w = f * 2^e into two parts, which can be
-// processed independently: An integral part p1, and a fractional part p2:
-//
-//      f * 2^e = ( (f div 2^-e) * 2^-e + (f mod 2^-e) ) * 2^e
-//              = (f div 2^-e) + (f mod 2^-e) * 2^e
-//              = p1 + p2 * 2^e
-//
-// The conversion of p1 into decimal form requires a series of divisions and
-// modulos by (a power of) 10. These operations are faster for 32-bit than for
-// 64-bit integers, so p1 should ideally fit into a 32-bit integer. This can be
-// achieved by choosing
-//
-//      -e >= 32   or   e <= -32 := gamma
-//
-// In order to convert the fractional part
-//
-//      p2 * 2^e = p2 / 2^-e = d[-1] / 10^1 + d[-2] / 10^2 + ...
-//
-// into decimal form, the fraction is repeatedly multiplied by 10 and the digits
-// d[-i] are extracted in order:
-//
-//      (10 * p2) div 2^-e = d[-1]
-//      (10 * p2) mod 2^-e = d[-2] / 10^1 + ...
-//
-// The multiplication by 10 must not overflow. It is sufficient to choose
-//
-//      10 * p2 < 16 * p2 = 2^4 * p2 <= 2^64.
-//
-// Since p2 = f mod 2^-e < 2^-e,
-//
-//      -e <= 60   or   e >= -60 := alpha
-
-constexpr int kAlpha = -60;
-constexpr int kGamma = -32;
+// Upper 128 bits of a 64 x 128 -> 192 bit multiplication.
+inline uint128 umul192_upper128(std::uint64_t x, uint128 y) noexcept {
+  uint128 r = umul128(x, y.high);
+  const std::uint64_t add = umul128_upper64(x, y.low);
+  const std::uint64_t sum = r.low + add;
+  r.high += (sum < r.low) ? 1 : 0;
+  r.low = sum;
+  return r;
+}
+
+// Lower 128 bits of a 64 x 128 -> 192 bit multiplication.
+inline uint128 umul192_lower128(std::uint64_t x, uint128 y) noexcept {
+  const std::uint64_t high = x * y.high;
+  const uint128 high_low = umul128(x, y.low);
+  return {high + high_low.high, high_low.low};
+}
+
+// Integer log approximations (exact over the range of inputs we feed them).
+inline int floor_log10_pow2(int e) noexcept { return (e * 315653) >> 20; }
+inline int floor_log2_pow10(int e) noexcept { return (e * 1741647) >> 19; }
+inline int floor_log10_pow2_minus_log10_4_over_3(int e) noexcept {
+  return (e * 631305 - 261663) >> 21;
+}
+
+// Format constants for IEEE-754 binary64, plus the precomputed cache of powers of ten.
+static constexpr int kappa = 2;
+static constexpr int significand_bits = 52;
+static constexpr int total_bits = 64;
+static constexpr int min_exponent = -1022;
+static constexpr int exponent_bias = -1023;
+static constexpr int cache_min_k = -292;
+static constexpr int big_divisor = 1000; // 10^(kappa + 1)
+static constexpr int small_divisor = 100; // 10^kappa
+static constexpr int case_shorter_interval_left_endpoint_lower_threshold = 2;
+static constexpr int case_shorter_interval_left_endpoint_upper_threshold = 3;
+static constexpr int shorter_interval_tie_lower_threshold = -77;
+static constexpr int shorter_interval_tie_upper_threshold = -77;
+
+// cache[i] holds a 128-bit approximation of a power of ten; indexed by
+// (-minus_k - cache_min_k). Taken verbatim from the Dragonbox reference.
+static constexpr uint128 cache[619] = {
+    {0xff77b1fcbebcdc4f, 0x25e8e89c13bb0f7b},
+    {0x9faacf3df73609b1, 0x77b191618c54e9ad},
+    {0xc795830d75038c1d, 0xd59df5b9ef6a2418},
+    {0xf97ae3d0d2446f25, 0x4b0573286b44ad1e},
+    {0x9becce62836ac577, 0x4ee367f9430aec33},
+    {0xc2e801fb244576d5, 0x229c41f793cda740},
+    {0xf3a20279ed56d48a, 0x6b43527578c11110},
+    {0x9845418c345644d6, 0x830a13896b78aaaa},
+    {0xbe5691ef416bd60c, 0x23cc986bc656d554},
+    {0xedec366b11c6cb8f, 0x2cbfbe86b7ec8aa9},
+    {0x94b3a202eb1c3f39, 0x7bf7d71432f3d6aa},
+    {0xb9e08a83a5e34f07, 0xdaf5ccd93fb0cc54},
+    {0xe858ad248f5c22c9, 0xd1b3400f8f9cff69},
+    {0x91376c36d99995be, 0x23100809b9c21fa2},
+    {0xb58547448ffffb2d, 0xabd40a0c2832a78b},
+    {0xe2e69915b3fff9f9, 0x16c90c8f323f516d},
+    {0x8dd01fad907ffc3b, 0xae3da7d97f6792e4},
+    {0xb1442798f49ffb4a, 0x99cd11cfdf41779d},
+    {0xdd95317f31c7fa1d, 0x40405643d711d584},
+    {0x8a7d3eef7f1cfc52, 0x482835ea666b2573},
+    {0xad1c8eab5ee43b66, 0xda3243650005eed0},
+    {0xd863b256369d4a40, 0x90bed43e40076a83},
+    {0x873e4f75e2224e68, 0x5a7744a6e804a292},
+    {0xa90de3535aaae202, 0x711515d0a205cb37},
+    {0xd3515c2831559a83, 0x0d5a5b44ca873e04},
+    {0x8412d9991ed58091, 0xe858790afe9486c3},
+    {0xa5178fff668ae0b6, 0x626e974dbe39a873},
+    {0xce5d73ff402d98e3, 0xfb0a3d212dc81290},
+    {0x80fa687f881c7f8e, 0x7ce66634bc9d0b9a},
+    {0xa139029f6a239f72, 0x1c1fffc1ebc44e81},
+    {0xc987434744ac874e, 0xa327ffb266b56221},
+    {0xfbe9141915d7a922, 0x4bf1ff9f0062baa9},
+    {0x9d71ac8fada6c9b5, 0x6f773fc3603db4aa},
+    {0xc4ce17b399107c22, 0xcb550fb4384d21d4},
+    {0xf6019da07f549b2b, 0x7e2a53a146606a49},
+    {0x99c102844f94e0fb, 0x2eda7444cbfc426e},
+    {0xc0314325637a1939, 0xfa911155fefb5309},
+    {0xf03d93eebc589f88, 0x793555ab7eba27cb},
+    {0x96267c7535b763b5, 0x4bc1558b2f3458df},
+    {0xbbb01b9283253ca2, 0x9eb1aaedfb016f17},
+    {0xea9c227723ee8bcb, 0x465e15a979c1cadd},
+    {0x92a1958a7675175f, 0x0bfacd89ec191eca},
+    {0xb749faed14125d36, 0xcef980ec671f667c},
+    {0xe51c79a85916f484, 0x82b7e12780e7401b},
+    {0x8f31cc0937ae58d2, 0xd1b2ecb8b0908811},
+    {0xb2fe3f0b8599ef07, 0x861fa7e6dcb4aa16},
+    {0xdfbdcece67006ac9, 0x67a791e093e1d49b},
+    {0x8bd6a141006042bd, 0xe0c8bb2c5c6d24e1},
+    {0xaecc49914078536d, 0x58fae9f773886e19},
+    {0xda7f5bf590966848, 0xaf39a475506a899f},
+    {0x888f99797a5e012d, 0x6d8406c952429604},
+    {0xaab37fd7d8f58178, 0xc8e5087ba6d33b84},
+    {0xd5605fcdcf32e1d6, 0xfb1e4a9a90880a65},
+    {0x855c3be0a17fcd26, 0x5cf2eea09a550680},
+    {0xa6b34ad8c9dfc06f, 0xf42faa48c0ea481f},
+    {0xd0601d8efc57b08b, 0xf13b94daf124da27},
+    {0x823c12795db6ce57, 0x76c53d08d6b70859},
+    {0xa2cb1717b52481ed, 0x54768c4b0c64ca6f},
+    {0xcb7ddcdda26da268, 0xa9942f5dcf7dfd0a},
+    {0xfe5d54150b090b02, 0xd3f93b35435d7c4d},
+    {0x9efa548d26e5a6e1, 0xc47bc5014a1a6db0},
+    {0xc6b8e9b0709f109a, 0x359ab6419ca1091c},
+    {0xf867241c8cc6d4c0, 0xc30163d203c94b63},
+    {0x9b407691d7fc44f8, 0x79e0de63425dcf1e},
+    {0xc21094364dfb5636, 0x985915fc12f542e5},
+    {0xf294b943e17a2bc4, 0x3e6f5b7b17b2939e},
+    {0x979cf3ca6cec5b5a, 0xa705992ceecf9c43},
+    {0xbd8430bd08277231, 0x50c6ff782a838354},
+    {0xece53cec4a314ebd, 0xa4f8bf5635246429},
+    {0x940f4613ae5ed136, 0x871b7795e136be9a},
+    {0xb913179899f68584, 0x28e2557b59846e40},
+    {0xe757dd7ec07426e5, 0x331aeada2fe589d0},
+    {0x9096ea6f3848984f, 0x3ff0d2c85def7622},
+    {0xb4bca50b065abe63, 0x0fed077a756b53aa},
+    {0xe1ebce4dc7f16dfb, 0xd3e8495912c62895},
+    {0x8d3360f09cf6e4bd, 0x64712dd7abbbd95d},
+    {0xb080392cc4349dec, 0xbd8d794d96aacfb4},
+    {0xdca04777f541c567, 0xecf0d7a0fc5583a1},
+    {0x89e42caaf9491b60, 0xf41686c49db57245},
+    {0xac5d37d5b79b6239, 0x311c2875c522ced6},
+    {0xd77485cb25823ac7, 0x7d633293366b828c},
+    {0x86a8d39ef77164bc, 0xae5dff9c02033198},
+    {0xa8530886b54dbdeb, 0xd9f57f830283fdfd},
+    {0xd267caa862a12d66, 0xd072df63c324fd7c},
+    {0x8380dea93da4bc60, 0x4247cb9e59f71e6e},
+    {0xa46116538d0deb78, 0x52d9be85f074e609},
+    {0xcd795be870516656, 0x67902e276c921f8c},
+    {0x806bd9714632dff6, 0x00ba1cd8a3db53b7},
+    {0xa086cfcd97bf97f3, 0x80e8a40eccd228a5},
+    {0xc8a883c0fdaf7df0, 0x6122cd128006b2ce},
+    {0xfad2a4b13d1b5d6c, 0x796b805720085f82},
+    {0x9cc3a6eec6311a63, 0xcbe3303674053bb1},
+    {0xc3f490aa77bd60fc, 0xbedbfc4411068a9d},
+    {0xf4f1b4d515acb93b, 0xee92fb5515482d45},
+    {0x991711052d8bf3c5, 0x751bdd152d4d1c4b},
+    {0xbf5cd54678eef0b6, 0xd262d45a78a0635e},
+    {0xef340a98172aace4, 0x86fb897116c87c35},
+    {0x9580869f0e7aac0e, 0xd45d35e6ae3d4da1},
+    {0xbae0a846d2195712, 0x8974836059cca10a},
+    {0xe998d258869facd7, 0x2bd1a438703fc94c},
+    {0x91ff83775423cc06, 0x7b6306a34627ddd0},
+    {0xb67f6455292cbf08, 0x1a3bc84c17b1d543},
+    {0xe41f3d6a7377eeca, 0x20caba5f1d9e4a94},
+    {0x8e938662882af53e, 0x547eb47b7282ee9d},
+    {0xb23867fb2a35b28d, 0xe99e619a4f23aa44},
+    {0xdec681f9f4c31f31, 0x6405fa00e2ec94d5},
+    {0x8b3c113c38f9f37e, 0xde83bc408dd3dd05},
+    {0xae0b158b4738705e, 0x9624ab50b148d446},
+    {0xd98ddaee19068c76, 0x3badd624dd9b0958},
+    {0x87f8a8d4cfa417c9, 0xe54ca5d70a80e5d7},
+    {0xa9f6d30a038d1dbc, 0x5e9fcf4ccd211f4d},
+    {0xd47487cc8470652b, 0x7647c32000696720},
+    {0x84c8d4dfd2c63f3b, 0x29ecd9f40041e074},
+    {0xa5fb0a17c777cf09, 0xf468107100525891},
+    {0xcf79cc9db955c2cc, 0x7182148d4066eeb5},
+    {0x81ac1fe293d599bf, 0xc6f14cd848405531},
+    {0xa21727db38cb002f, 0xb8ada00e5a506a7d},
+    {0xca9cf1d206fdc03b, 0xa6d90811f0e4851d},
+    {0xfd442e4688bd304a, 0x908f4a166d1da664},
+    {0x9e4a9cec15763e2e, 0x9a598e4e043287ff},
+    {0xc5dd44271ad3cdba, 0x40eff1e1853f29fe},
+    {0xf7549530e188c128, 0xd12bee59e68ef47d},
+    {0x9a94dd3e8cf578b9, 0x82bb74f8301958cf},
+    {0xc13a148e3032d6e7, 0xe36a52363c1faf02},
+    {0xf18899b1bc3f8ca1, 0xdc44e6c3cb279ac2},
+    {0x96f5600f15a7b7e5, 0x29ab103a5ef8c0ba},
+    {0xbcb2b812db11a5de, 0x7415d448f6b6f0e8},
+    {0xebdf661791d60f56, 0x111b495b3464ad22},
+    {0x936b9fcebb25c995, 0xcab10dd900beec35},
+    {0xb84687c269ef3bfb, 0x3d5d514f40eea743},
+    {0xe65829b3046b0afa, 0x0cb4a5a3112a5113},
+    {0x8ff71a0fe2c2e6dc, 0x47f0e785eaba72ac},
+    {0xb3f4e093db73a093, 0x59ed216765690f57},
+    {0xe0f218b8d25088b8, 0x306869c13ec3532d},
+    {0x8c974f7383725573, 0x1e414218c73a13fc},
+    {0xafbd2350644eeacf, 0xe5d1929ef90898fb},
+    {0xdbac6c247d62a583, 0xdf45f746b74abf3a},
+    {0x894bc396ce5da772, 0x6b8bba8c328eb784},
+    {0xab9eb47c81f5114f, 0x066ea92f3f326565},
+    {0xd686619ba27255a2, 0xc80a537b0efefebe},
+    {0x8613fd0145877585, 0xbd06742ce95f5f37},
+    {0xa798fc4196e952e7, 0x2c48113823b73705},
+    {0xd17f3b51fca3a7a0, 0xf75a15862ca504c6},
+    {0x82ef85133de648c4, 0x9a984d73dbe722fc},
+    {0xa3ab66580d5fdaf5, 0xc13e60d0d2e0ebbb},
+    {0xcc963fee10b7d1b3, 0x318df905079926a9},
+    {0xffbbcfe994e5c61f, 0xfdf17746497f7053},
+    {0x9fd561f1fd0f9bd3, 0xfeb6ea8bedefa634},
+    {0xc7caba6e7c5382c8, 0xfe64a52ee96b8fc1},
+    {0xf9bd690a1b68637b, 0x3dfdce7aa3c673b1},
+    {0x9c1661a651213e2d, 0x06bea10ca65c084f},
+    {0xc31bfa0fe5698db8, 0x486e494fcff30a63},
+    {0xf3e2f893dec3f126, 0x5a89dba3c3efccfb},
+    {0x986ddb5c6b3a76b7, 0xf89629465a75e01d},
+    {0xbe89523386091465, 0xf6bbb397f1135824},
+    {0xee2ba6c0678b597f, 0x746aa07ded582e2d},
+    {0x94db483840b717ef, 0xa8c2a44eb4571cdd},
+    {0xba121a4650e4ddeb, 0x92f34d62616ce414},
+    {0xe896a0d7e51e1566, 0x77b020baf9c81d18},
+    {0x915e2486ef32cd60, 0x0ace1474dc1d122f},
+    {0xb5b5ada8aaff80b8, 0x0d819992132456bb},
+    {0xe3231912d5bf60e6, 0x10e1fff697ed6c6a},
+    {0x8df5efabc5979c8f, 0xca8d3ffa1ef463c2},
+    {0xb1736b96b6fd83b3, 0xbd308ff8a6b17cb3},
+    {0xddd0467c64bce4a0, 0xac7cb3f6d05ddbdf},
+    {0x8aa22c0dbef60ee4, 0x6bcdf07a423aa96c},
+    {0xad4ab7112eb3929d, 0x86c16c98d2c953c7},
+    {0xd89d64d57a607744, 0xe871c7bf077ba8b8},
+    {0x87625f056c7c4a8b, 0x11471cd764ad4973},
+    {0xa93af6c6c79b5d2d, 0xd598e40d3dd89bd0},
+    {0xd389b47879823479, 0x4aff1d108d4ec2c4},
+    {0x843610cb4bf160cb, 0xcedf722a585139bb},
+    {0xa54394fe1eedb8fe, 0xc2974eb4ee658829},
+    {0xce947a3da6a9273e, 0x733d226229feea33},
+    {0x811ccc668829b887, 0x0806357d5a3f5260},
+    {0xa163ff802a3426a8, 0xca07c2dcb0cf26f8},
+    {0xc9bcff6034c13052, 0xfc89b393dd02f0b6},
+    {0xfc2c3f3841f17c67, 0xbbac2078d443ace3},
+    {0x9d9ba7832936edc0, 0xd54b944b84aa4c0e},
+    {0xc5029163f384a931, 0x0a9e795e65d4df12},
+    {0xf64335bcf065d37d, 0x4d4617b5ff4a16d6},
+    {0x99ea0196163fa42e, 0x504bced1bf8e4e46},
+    {0xc06481fb9bcf8d39, 0xe45ec2862f71e1d7},
+    {0xf07da27a82c37088, 0x5d767327bb4e5a4d},
+    {0x964e858c91ba2655, 0x3a6a07f8d510f870},
+    {0xbbe226efb628afea, 0x890489f70a55368c},
+    {0xeadab0aba3b2dbe5, 0x2b45ac74ccea842f},
+    {0x92c8ae6b464fc96f, 0x3b0b8bc90012929e},
+    {0xb77ada0617e3bbcb, 0x09ce6ebb40173745},
+    {0xe55990879ddcaabd, 0xcc420a6a101d0516},
+    {0x8f57fa54c2a9eab6, 0x9fa946824a12232e},
+    {0xb32df8e9f3546564, 0x47939822dc96abfa},
+    {0xdff9772470297ebd, 0x59787e2b93bc56f8},
+    {0x8bfbea76c619ef36, 0x57eb4edb3c55b65b},
+    {0xaefae51477a06b03, 0xede622920b6b23f2},
+    {0xdab99e59958885c4, 0xe95fab368e45ecee},
+    {0x88b402f7fd75539b, 0x11dbcb0218ebb415},
+    {0xaae103b5fcd2a881, 0xd652bdc29f26a11a},
+    {0xd59944a37c0752a2, 0x4be76d3346f04960},
+    {0x857fcae62d8493a5, 0x6f70a4400c562ddc},
+    {0xa6dfbd9fb8e5b88e, 0xcb4ccd500f6bb953},
+    {0xd097ad07a71f26b2, 0x7e2000a41346a7a8},
+    {0x825ecc24c873782f, 0x8ed400668c0c28c9},
+    {0xa2f67f2dfa90563b, 0x728900802f0f32fb},
+    {0xcbb41ef979346bca, 0x4f2b40a03ad2ffba},
+    {0xfea126b7d78186bc, 0xe2f610c84987bfa9},
+    {0x9f24b832e6b0f436, 0x0dd9ca7d2df4d7ca},
+    {0xc6ede63fa05d3143, 0x91503d1c79720dbc},
+    {0xf8a95fcf88747d94, 0x75a44c6397ce912b},
+    {0x9b69dbe1b548ce7c, 0xc986afbe3ee11abb},
+    {0xc24452da229b021b, 0xfbe85badce996169},
+    {0xf2d56790ab41c2a2, 0xfae27299423fb9c4},
+    {0x97c560ba6b0919a5, 0xdccd879fc967d41b},
+    {0xbdb6b8e905cb600f, 0x5400e987bbc1c921},
+    {0xed246723473e3813, 0x290123e9aab23b69},
+    {0x9436c0760c86e30b, 0xf9a0b6720aaf6522},
+    {0xb94470938fa89bce, 0xf808e40e8d5b3e6a},
+    {0xe7958cb87392c2c2, 0xb60b1d1230b20e05},
+    {0x90bd77f3483bb9b9, 0xb1c6f22b5e6f48c3},
+    {0xb4ecd5f01a4aa828, 0x1e38aeb6360b1af4},
+    {0xe2280b6c20dd5232, 0x25c6da63c38de1b1},
+    {0x8d590723948a535f, 0x579c487e5a38ad0f},
+    {0xb0af48ec79ace837, 0x2d835a9df0c6d852},
+    {0xdcdb1b2798182244, 0xf8e431456cf88e66},
+    {0x8a08f0f8bf0f156b, 0x1b8e9ecb641b5900},
+    {0xac8b2d36eed2dac5, 0xe272467e3d222f40},
+    {0xd7adf884aa879177, 0x5b0ed81dcc6abb10},
+    {0x86ccbb52ea94baea, 0x98e947129fc2b4ea},
+    {0xa87fea27a539e9a5, 0x3f2398d747b36225},
+    {0xd29fe4b18e88640e, 0x8eec7f0d19a03aae},
+    {0x83a3eeeef9153e89, 0x1953cf68300424ad},
+    {0xa48ceaaab75a8e2b, 0x5fa8c3423c052dd8},
+    {0xcdb02555653131b6, 0x3792f412cb06794e},
+    {0x808e17555f3ebf11, 0xe2bbd88bbee40bd1},
+    {0xa0b19d2ab70e6ed6, 0x5b6aceaeae9d0ec5},
+    {0xc8de047564d20a8b, 0xf245825a5a445276},
+    {0xfb158592be068d2e, 0xeed6e2f0f0d56713},
+    {0x9ced737bb6c4183d, 0x55464dd69685606c},
+    {0xc428d05aa4751e4c, 0xaa97e14c3c26b887},
+    {0xf53304714d9265df, 0xd53dd99f4b3066a9},
+    {0x993fe2c6d07b7fab, 0xe546a8038efe402a},
+    {0xbf8fdb78849a5f96, 0xde98520472bdd034},
+    {0xef73d256a5c0f77c, 0x963e66858f6d4441},
+    {0x95a8637627989aad, 0xdde7001379a44aa9},
+    {0xbb127c53b17ec159, 0x5560c018580d5d53},
+    {0xe9d71b689dde71af, 0xaab8f01e6e10b4a7},
+    {0x9226712162ab070d, 0xcab3961304ca70e9},
+    {0xb6b00d69bb55c8d1, 0x3d607b97c5fd0d23},
+    {0xe45c10c42a2b3b05, 0x8cb89a7db77c506b},
+    {0x8eb98a7a9a5b04e3, 0x77f3608e92adb243},
+    {0xb267ed1940f1c61c, 0x55f038b237591ed4},
+    {0xdf01e85f912e37a3, 0x6b6c46dec52f6689},
+    {0x8b61313bbabce2c6, 0x2323ac4b3b3da016},
+    {0xae397d8aa96c1b77, 0xabec975e0a0d081b},
+    {0xd9c7dced53c72255, 0x96e7bd358c904a22},
+    {0x881cea14545c7575, 0x7e50d64177da2e55},
+    {0xaa242499697392d2, 0xdde50bd1d5d0b9ea},
+    {0xd4ad2dbfc3d07787, 0x955e4ec64b44e865},
+    {0x84ec3c97da624ab4, 0xbd5af13bef0b113f},
+    {0xa6274bbdd0fadd61, 0xecb1ad8aeacdd58f},
+    {0xcfb11ead453994ba, 0x67de18eda5814af3},
+    {0x81ceb32c4b43fcf4, 0x80eacf948770ced8},
+    {0xa2425ff75e14fc31, 0xa1258379a94d028e},
+    {0xcad2f7f5359a3b3e, 0x096ee45813a04331},
+    {0xfd87b5f28300ca0d, 0x8bca9d6e188853fd},
+    {0x9e74d1b791e07e48, 0x775ea264cf55347e},
+    {0xc612062576589dda, 0x95364afe032a819e},
+    {0xf79687aed3eec551, 0x3a83ddbd83f52205},
+    {0x9abe14cd44753b52, 0xc4926a9672793543},
+    {0xc16d9a0095928a27, 0x75b7053c0f178294},
+    {0xf1c90080baf72cb1, 0x5324c68b12dd6339},
+    {0x971da05074da7bee, 0xd3f6fc16ebca5e04},
+    {0xbce5086492111aea, 0x88f4bb1ca6bcf585},
+    {0xec1e4a7db69561a5, 0x2b31e9e3d06c32e6},
+    {0x9392ee8e921d5d07, 0x3aff322e62439fd0},
+    {0xb877aa3236a4b449, 0x09befeb9fad487c3},
+    {0xe69594bec44de15b, 0x4c2ebe687989a9b4},
+    {0x901d7cf73ab0acd9, 0x0f9d37014bf60a11},
+    {0xb424dc35095cd80f, 0x538484c19ef38c95},
+    {0xe12e13424bb40e13, 0x2865a5f206b06fba},
+    {0x8cbccc096f5088cb, 0xf93f87b7442e45d4},
+    {0xafebff0bcb24aafe, 0xf78f69a51539d749},
+    {0xdbe6fecebdedd5be, 0xb573440e5a884d1c},
+    {0x89705f4136b4a597, 0x31680a88f8953031},
+    {0xabcc77118461cefc, 0xfdc20d2b36ba7c3e},
+    {0xd6bf94d5e57a42bc, 0x3d32907604691b4d},
+    {0x8637bd05af6c69b5, 0xa63f9a49c2c1b110},
+    {0xa7c5ac471b478423, 0x0fcf80dc33721d54},
+    {0xd1b71758e219652b, 0xd3c36113404ea4a9},
+    {0x83126e978d4fdf3b, 0x645a1cac083126ea},
+    {0xa3d70a3d70a3d70a, 0x3d70a3d70a3d70a4},
+    {0xcccccccccccccccc, 0xcccccccccccccccd},
+    {0x8000000000000000, 0x0000000000000000},
+    {0xa000000000000000, 0x0000000000000000},
+    {0xc800000000000000, 0x0000000000000000},
+    {0xfa00000000000000, 0x0000000000000000},
+    {0x9c40000000000000, 0x0000000000000000},
+    {0xc350000000000000, 0x0000000000000000},
+    {0xf424000000000000, 0x0000000000000000},
+    {0x9896800000000000, 0x0000000000000000},
+    {0xbebc200000000000, 0x0000000000000000},
+    {0xee6b280000000000, 0x0000000000000000},
+    {0x9502f90000000000, 0x0000000000000000},
+    {0xba43b74000000000, 0x0000000000000000},
+    {0xe8d4a51000000000, 0x0000000000000000},
+    {0x9184e72a00000000, 0x0000000000000000},
+    {0xb5e620f480000000, 0x0000000000000000},
+    {0xe35fa931a0000000, 0x0000000000000000},
+    {0x8e1bc9bf04000000, 0x0000000000000000},
+    {0xb1a2bc2ec5000000, 0x0000000000000000},
+    {0xde0b6b3a76400000, 0x0000000000000000},
+    {0x8ac7230489e80000, 0x0000000000000000},
+    {0xad78ebc5ac620000, 0x0000000000000000},
+    {0xd8d726b7177a8000, 0x0000000000000000},
+    {0x878678326eac9000, 0x0000000000000000},
+    {0xa968163f0a57b400, 0x0000000000000000},
+    {0xd3c21bcecceda100, 0x0000000000000000},
+    {0x84595161401484a0, 0x0000000000000000},
+    {0xa56fa5b99019a5c8, 0x0000000000000000},
+    {0xcecb8f27f4200f3a, 0x0000000000000000},
+    {0x813f3978f8940984, 0x4000000000000000},
+    {0xa18f07d736b90be5, 0x5000000000000000},
+    {0xc9f2c9cd04674ede, 0xa400000000000000},
+    {0xfc6f7c4045812296, 0x4d00000000000000},
+    {0x9dc5ada82b70b59d, 0xf020000000000000},
+    {0xc5371912364ce305, 0x6c28000000000000},
+    {0xf684df56c3e01bc6, 0xc732000000000000},
+    {0x9a130b963a6c115c, 0x3c7f400000000000},
+    {0xc097ce7bc90715b3, 0x4b9f100000000000},
+    {0xf0bdc21abb48db20, 0x1e86d40000000000},
+    {0x96769950b50d88f4, 0x1314448000000000},
+    {0xbc143fa4e250eb31, 0x17d955a000000000},
+    {0xeb194f8e1ae525fd, 0x5dcfab0800000000},
+    {0x92efd1b8d0cf37be, 0x5aa1cae500000000},
+    {0xb7abc627050305ad, 0xf14a3d9e40000000},
+    {0xe596b7b0c643c719, 0x6d9ccd05d0000000},
+    {0x8f7e32ce7bea5c6f, 0xe4820023a2000000},
+    {0xb35dbf821ae4f38b, 0xdda2802c8a800000},
+    {0xe0352f62a19e306e, 0xd50b2037ad200000},
+    {0x8c213d9da502de45, 0x4526f422cc340000},
+    {0xaf298d050e4395d6, 0x9670b12b7f410000},
+    {0xdaf3f04651d47b4c, 0x3c0cdd765f114000},
+    {0x88d8762bf324cd0f, 0xa5880a69fb6ac800},
+    {0xab0e93b6efee0053, 0x8eea0d047a457a00},
+    {0xd5d238a4abe98068, 0x72a4904598d6d880},
+    {0x85a36366eb71f041, 0x47a6da2b7f864750},
+    {0xa70c3c40a64e6c51, 0x999090b65f67d924},
+    {0xd0cf4b50cfe20765, 0xfff4b4e3f741cf6d},
+    {0x82818f1281ed449f, 0xbff8f10e7a8921a5},
+    {0xa321f2d7226895c7, 0xaff72d52192b6a0e},
+    {0xcbea6f8ceb02bb39, 0x9bf4f8a69f764491},
+    {0xfee50b7025c36a08, 0x02f236d04753d5b5},
+    {0x9f4f2726179a2245, 0x01d762422c946591},
+    {0xc722f0ef9d80aad6, 0x424d3ad2b7b97ef6},
+    {0xf8ebad2b84e0d58b, 0xd2e0898765a7deb3},
+    {0x9b934c3b330c8577, 0x63cc55f49f88eb30},
+    {0xc2781f49ffcfa6d5, 0x3cbf6b71c76b25fc},
+    {0xf316271c7fc3908a, 0x8bef464e3945ef7b},
+    {0x97edd871cfda3a56, 0x97758bf0e3cbb5ad},
+    {0xbde94e8e43d0c8ec, 0x3d52eeed1cbea318},
+    {0xed63a231d4c4fb27, 0x4ca7aaa863ee4bde},
+    {0x945e455f24fb1cf8, 0x8fe8caa93e74ef6b},
+    {0xb975d6b6ee39e436, 0xb3e2fd538e122b45},
+    {0xe7d34c64a9c85d44, 0x60dbbca87196b617},
+    {0x90e40fbeea1d3a4a, 0xbc8955e946fe31ce},
+    {0xb51d13aea4a488dd, 0x6babab6398bdbe42},
+    {0xe264589a4dcdab14, 0xc696963c7eed2dd2},
+    {0x8d7eb76070a08aec, 0xfc1e1de5cf543ca3},
+    {0xb0de65388cc8ada8, 0x3b25a55f43294bcc},
+    {0xdd15fe86affad912, 0x49ef0eb713f39ebf},
+    {0x8a2dbf142dfcc7ab, 0x6e3569326c784338},
+    {0xacb92ed9397bf996, 0x49c2c37f07965405},
+    {0xd7e77a8f87daf7fb, 0xdc33745ec97be907},
+    {0x86f0ac99b4e8dafd, 0x69a028bb3ded71a4},
+    {0xa8acd7c0222311bc, 0xc40832ea0d68ce0d},
+    {0xd2d80db02aabd62b, 0xf50a3fa490c30191},
+    {0x83c7088e1aab65db, 0x792667c6da79e0fb},
+    {0xa4b8cab1a1563f52, 0x577001b891185939},
+    {0xcde6fd5e09abcf26, 0xed4c0226b55e6f87},
+    {0x80b05e5ac60b6178, 0x544f8158315b05b5},
+    {0xa0dc75f1778e39d6, 0x696361ae3db1c722},
+    {0xc913936dd571c84c, 0x03bc3a19cd1e38ea},
+    {0xfb5878494ace3a5f, 0x04ab48a04065c724},
+    {0x9d174b2dcec0e47b, 0x62eb0d64283f9c77},
+    {0xc45d1df942711d9a, 0x3ba5d0bd324f8395},
+    {0xf5746577930d6500, 0xca8f44ec7ee3647a},
+    {0x9968bf6abbe85f20, 0x7e998b13cf4e1ecc},
+    {0xbfc2ef456ae276e8, 0x9e3fedd8c321a67f},
+    {0xefb3ab16c59b14a2, 0xc5cfe94ef3ea101f},
+    {0x95d04aee3b80ece5, 0xbba1f1d158724a13},
+    {0xbb445da9ca61281f, 0x2a8a6e45ae8edc98},
+    {0xea1575143cf97226, 0xf52d09d71a3293be},
+    {0x924d692ca61be758, 0x593c2626705f9c57},
+    {0xb6e0c377cfa2e12e, 0x6f8b2fb00c77836d},
+    {0xe498f455c38b997a, 0x0b6dfb9c0f956448},
+    {0x8edf98b59a373fec, 0x4724bd4189bd5ead},
+    {0xb2977ee300c50fe7, 0x58edec91ec2cb658},
+    {0xdf3d5e9bc0f653e1, 0x2f2967b66737e3ee},
+    {0x8b865b215899f46c, 0xbd79e0d20082ee75},
+    {0xae67f1e9aec07187, 0xecd8590680a3aa12},
+    {0xda01ee641a708de9, 0xe80e6f4820cc9496},
+    {0x884134fe908658b2, 0x3109058d147fdcde},
+    {0xaa51823e34a7eede, 0xbd4b46f0599fd416},
+    {0xd4e5e2cdc1d1ea96, 0x6c9e18ac7007c91b},
+    {0x850fadc09923329e, 0x03e2cf6bc604ddb1},
+    {0xa6539930bf6bff45, 0x84db8346b786151d},
+    {0xcfe87f7cef46ff16, 0xe612641865679a64},
+    {0x81f14fae158c5f6e, 0x4fcb7e8f3f60c07f},
+    {0xa26da3999aef7749, 0xe3be5e330f38f09e},
+    {0xcb090c8001ab551c, 0x5cadf5bfd3072cc6},
+    {0xfdcb4fa002162a63, 0x73d9732fc7c8f7f7},
+    {0x9e9f11c4014dda7e, 0x2867e7fddcdd9afb},
+    {0xc646d63501a1511d, 0xb281e1fd541501b9},
+    {0xf7d88bc24209a565, 0x1f225a7ca91a4227},
+    {0x9ae757596946075f, 0x3375788de9b06959},
+    {0xc1a12d2fc3978937, 0x0052d6b1641c83af},
+    {0xf209787bb47d6b84, 0xc0678c5dbd23a49b},
+    {0x9745eb4d50ce6332, 0xf840b7ba963646e1},
+    {0xbd176620a501fbff, 0xb650e5a93bc3d899},
+    {0xec5d3fa8ce427aff, 0xa3e51f138ab4cebf},
+    {0x93ba47c980e98cdf, 0xc66f336c36b10138},
+    {0xb8a8d9bbe123f017, 0xb80b0047445d4185},
+    {0xe6d3102ad96cec1d, 0xa60dc059157491e6},
+    {0x9043ea1ac7e41392, 0x87c89837ad68db30},
+    {0xb454e4a179dd1877, 0x29babe4598c311fc},
+    {0xe16a1dc9d8545e94, 0xf4296dd6fef3d67b},
+    {0x8ce2529e2734bb1d, 0x1899e4a65f58660d},
+    {0xb01ae745b101e9e4, 0x5ec05dcff72e7f90},
+    {0xdc21a1171d42645d, 0x76707543f4fa1f74},
+    {0x899504ae72497eba, 0x6a06494a791c53a9},
+    {0xabfa45da0edbde69, 0x0487db9d17636893},
+    {0xd6f8d7509292d603, 0x45a9d2845d3c42b7},
+    {0x865b86925b9bc5c2, 0x0b8a2392ba45a9b3},
+    {0xa7f26836f282b732, 0x8e6cac7768d7141f},
+    {0xd1ef0244af2364ff, 0x3207d795430cd927},
+    {0x8335616aed761f1f, 0x7f44e6bd49e807b9},
+    {0xa402b9c5a8d3a6e7, 0x5f16206c9c6209a7},
+    {0xcd036837130890a1, 0x36dba887c37a8c10},
+    {0x802221226be55a64, 0xc2494954da2c978a},
+    {0xa02aa96b06deb0fd, 0xf2db9baa10b7bd6d},
+    {0xc83553c5c8965d3d, 0x6f92829494e5acc8},
+    {0xfa42a8b73abbf48c, 0xcb772339ba1f17fa},
+    {0x9c69a97284b578d7, 0xff2a760414536efc},
+    {0xc38413cf25e2d70d, 0xfef5138519684abb},
+    {0xf46518c2ef5b8cd1, 0x7eb258665fc25d6a},
+    {0x98bf2f79d5993802, 0xef2f773ffbd97a62},
+    {0xbeeefb584aff8603, 0xaafb550ffacfd8fb},
+    {0xeeaaba2e5dbf6784, 0x95ba2a53f983cf39},
+    {0x952ab45cfa97a0b2, 0xdd945a747bf26184},
+    {0xba756174393d88df, 0x94f971119aeef9e5},
+    {0xe912b9d1478ceb17, 0x7a37cd5601aab85e},
+    {0x91abb422ccb812ee, 0xac62e055c10ab33b},
+    {0xb616a12b7fe617aa, 0x577b986b314d600a},
+    {0xe39c49765fdf9d94, 0xed5a7e85fda0b80c},
+    {0x8e41ade9fbebc27d, 0x14588f13be847308},
+    {0xb1d219647ae6b31c, 0x596eb2d8ae258fc9},
+    {0xde469fbd99a05fe3, 0x6fca5f8ed9aef3bc},
+    {0x8aec23d680043bee, 0x25de7bb9480d5855},
+    {0xada72ccc20054ae9, 0xaf561aa79a10ae6b},
+    {0xd910f7ff28069da4, 0x1b2ba1518094da05},
+    {0x87aa9aff79042286, 0x90fb44d2f05d0843},
+    {0xa99541bf57452b28, 0x353a1607ac744a54},
+    {0xd3fa922f2d1675f2, 0x42889b8997915ce9},
+    {0x847c9b5d7c2e09b7, 0x69956135febada12},
+    {0xa59bc234db398c25, 0x43fab9837e699096},
+    {0xcf02b2c21207ef2e, 0x94f967e45e03f4bc},
+    {0x8161afb94b44f57d, 0x1d1be0eebac278f6},
+    {0xa1ba1ba79e1632dc, 0x6462d92a69731733},
+    {0xca28a291859bbf93, 0x7d7b8f7503cfdcff},
+    {0xfcb2cb35e702af78, 0x5cda735244c3d43f},
+    {0x9defbf01b061adab, 0x3a0888136afa64a8},
+    {0xc56baec21c7a1916, 0x088aaa1845b8fdd1},
+    {0xf6c69a72a3989f5b, 0x8aad549e57273d46},
+    {0x9a3c2087a63f6399, 0x36ac54e2f678864c},
+    {0xc0cb28a98fcf3c7f, 0x84576a1bb416a7de},
+    {0xf0fdf2d3f3c30b9f, 0x656d44a2a11c51d6},
+    {0x969eb7c47859e743, 0x9f644ae5a4b1b326},
+    {0xbc4665b596706114, 0x873d5d9f0dde1fef},
+    {0xeb57ff22fc0c7959, 0xa90cb506d155a7eb},
+    {0x9316ff75dd87cbd8, 0x09a7f12442d588f3},
+    {0xb7dcbf5354e9bece, 0x0c11ed6d538aeb30},
+    {0xe5d3ef282a242e81, 0x8f1668c8a86da5fb},
+    {0x8fa475791a569d10, 0xf96e017d694487bd},
+    {0xb38d92d760ec4455, 0x37c981dcc395a9ad},
+    {0xe070f78d3927556a, 0x85bbe253f47b1418},
+    {0x8c469ab843b89562, 0x93956d7478ccec8f},
+    {0xaf58416654a6babb, 0x387ac8d1970027b3},
+    {0xdb2e51bfe9d0696a, 0x06997b05fcc0319f},
+    {0x88fcf317f22241e2, 0x441fece3bdf81f04},
+    {0xab3c2fddeeaad25a, 0xd527e81cad7626c4},
+    {0xd60b3bd56a5586f1, 0x8a71e223d8d3b075},
+    {0x85c7056562757456, 0xf6872d5667844e4a},
+    {0xa738c6bebb12d16c, 0xb428f8ac016561dc},
+    {0xd106f86e69d785c7, 0xe13336d701beba53},
+    {0x82a45b450226b39c, 0xecc0024661173474},
+    {0xa34d721642b06084, 0x27f002d7f95d0191},
+    {0xcc20ce9bd35c78a5, 0x31ec038df7b441f5},
+    {0xff290242c83396ce, 0x7e67047175a15272},
+    {0x9f79a169bd203e41, 0x0f0062c6e984d387},
+    {0xc75809c42c684dd1, 0x52c07b78a3e60869},
+    {0xf92e0c3537826145, 0xa7709a56ccdf8a83},
+    {0x9bbcc7a142b17ccb, 0x88a66076400bb692},
+    {0xc2abf989935ddbfe, 0x6acff893d00ea436},
+    {0xf356f7ebf83552fe, 0x0583f6b8c4124d44},
+    {0x98165af37b2153de, 0xc3727a337a8b704b},
+    {0xbe1bf1b059e9a8d6, 0x744f18c0592e4c5d},
+    {0xeda2ee1c7064130c, 0x1162def06f79df74},
+    {0x9485d4d1c63e8be7, 0x8addcb5645ac2ba9},
+    {0xb9a74a0637ce2ee1, 0x6d953e2bd7173693},
+    {0xe8111c87c5c1ba99, 0xc8fa8db6ccdd0438},
+    {0x910ab1d4db9914a0, 0x1d9c9892400a22a3},
+    {0xb54d5e4a127f59c8, 0x2503beb6d00cab4c},
+    {0xe2a0b5dc971f303a, 0x2e44ae64840fd61e},
+    {0x8da471a9de737e24, 0x5ceaecfed289e5d3},
+    {0xb10d8e1456105dad, 0x7425a83e872c5f48},
+    {0xdd50f1996b947518, 0xd12f124e28f7771a},
+    {0x8a5296ffe33cc92f, 0x82bd6b70d99aaa70},
+    {0xace73cbfdc0bfb7b, 0x636cc64d1001550c},
+    {0xd8210befd30efa5a, 0x3c47f7e05401aa4f},
+    {0x8714a775e3e95c78, 0x65acfaec34810a72},
+    {0xa8d9d1535ce3b396, 0x7f1839a741a14d0e},
+    {0xd31045a8341ca07c, 0x1ede48111209a051},
+    {0x83ea2b892091e44d, 0x934aed0aab460433},
+    {0xa4e4b66b68b65d60, 0xf81da84d56178540},
+    {0xce1de40642e3f4b9, 0x36251260ab9d668f},
+    {0x80d2ae83e9ce78f3, 0xc1d72b7c6b42601a},
+    {0xa1075a24e4421730, 0xb24cf65b8612f820},
+    {0xc94930ae1d529cfc, 0xdee033f26797b628},
+    {0xfb9b7cd9a4a7443c, 0x169840ef017da3b2},
+    {0x9d412e0806e88aa5, 0x8e1f289560ee864f},
+    {0xc491798a08a2ad4e, 0xf1a6f2bab92a27e3},
+    {0xf5b5d7ec8acb58a2, 0xae10af696774b1dc},
+    {0x9991a6f3d6bf1765, 0xacca6da1e0a8ef2a},
+    {0xbff610b0cc6edd3f, 0x17fd090a58d32af4},
+    {0xeff394dcff8a948e, 0xddfc4b4cef07f5b1},
+    {0x95f83d0a1fb69cd9, 0x4abdaf101564f98f},
+    {0xbb764c4ca7a4440f, 0x9d6d1ad41abe37f2},
+    {0xea53df5fd18d5513, 0x84c86189216dc5ee},
+    {0x92746b9be2f8552c, 0x32fd3cf5b4e49bb5},
+    {0xb7118682dbb66a77, 0x3fbc8c33221dc2a2},
+    {0xe4d5e82392a40515, 0x0fabaf3feaa5334b},
+    {0x8f05b1163ba6832d, 0x29cb4d87f2a7400f},
+    {0xb2c71d5bca9023f8, 0x743e20e9ef511013},
+    {0xdf78e4b2bd342cf6, 0x914da9246b255417},
+    {0x8bab8eefb6409c1a, 0x1ad089b6c2f7548f},
+    {0xae9672aba3d0c320, 0xa184ac2473b529b2},
+    {0xda3c0f568cc4f3e8, 0xc9e5d72d90a2741f},
+    {0x8865899617fb1871, 0x7e2fa67c7a658893},
+    {0xaa7eebfb9df9de8d, 0xddbb901b98feeab8},
+    {0xd51ea6fa85785631, 0x552a74227f3ea566},
+    {0x8533285c936b35de, 0xd53a88958f872760},
+    {0xa67ff273b8460356, 0x8a892abaf368f138},
+    {0xd01fef10a657842c, 0x2d2b7569b0432d86},
+    {0x8213f56a67f6b29b, 0x9c3b29620e29fc74},
+    {0xa298f2c501f45f42, 0x8349f3ba91b47b90},
+    {0xcb3f2f7642717713, 0x241c70a936219a74},
+    {0xfe0efb53d30dd4d7, 0xed238cd383aa0111},
+    {0x9ec95d1463e8a506, 0xf4363804324a40ab},
+    {0xc67bb4597ce2ce48, 0xb143c6053edcd0d6},
+    {0xf81aa16fdc1b81da, 0xdd94b7868e94050b},
+    {0x9b10a4e5e9913128, 0xca7cf2b4191c8327},
+    {0xc1d4ce1f63f57d72, 0xfd1c2f611f63a3f1},
+    {0xf24a01a73cf2dccf, 0xbc633b39673c8ced},
+    {0x976e41088617ca01, 0xd5be0503e085d814},
+    {0xbd49d14aa79dbc82, 0x4b2d8644d8a74e19},
+    {0xec9c459d51852ba2, 0xddf8e7d60ed1219f},
+    {0x93e1ab8252f33b45, 0xcabb90e5c942b504},
+    {0xb8da1662e7b00a17, 0x3d6a751f3b936244},
+    {0xe7109bfba19c0c9d, 0x0cc512670a783ad5},
+    {0x906a617d450187e2, 0x27fb2b80668b24c6},
+    {0xb484f9dc9641e9da, 0xb1f9f660802dedf7},
+    {0xe1a63853bbd26451, 0x5e7873f8a0396974},
+    {0x8d07e33455637eb2, 0xdb0b487b6423e1e9},
+    {0xb049dc016abc5e5f, 0x91ce1a9a3d2cda63},
+    {0xdc5c5301c56b75f7, 0x7641a140cc7810fc},
+    {0x89b9b3e11b6329ba, 0xa9e904c87fcb0a9e},
+    {0xac2820d9623bf429, 0x546345fa9fbdcd45},
+    {0xd732290fbacaf133, 0xa97c177947ad4096},
+    {0x867f59a9d4bed6c0, 0x49ed8eabcccc485e},
+    {0xa81f301449ee8c70, 0x5c68f256bfff5a75},
+    {0xd226fc195c6a2f8c, 0x73832eec6fff3112},
+    {0x83585d8fd9c25db7, 0xc831fd53c5ff7eac},
+    {0xa42e74f3d032f525, 0xba3e7ca8b77f5e56},
+    {0xcd3a1230c43fb26f, 0x28ce1bd2e55f35ec},
+    {0x80444b5e7aa7cf85, 0x7980d163cf5b81b4},
+    {0xa0555e361951c366, 0xd7e105bcc3326220},
+    {0xc86ab5c39fa63440, 0x8dd9472bf3fefaa8},
+    {0xfa856334878fc150, 0xb14f98f6f0feb952},
+    {0x9c935e00d4b9d8d2, 0x6ed1bf9a569f33d4},
+    {0xc3b8358109e84f07, 0x0a862f80ec4700c9},
+    {0xf4a642e14c6262c8, 0xcd27bb612758c0fb},
+    {0x98e7e9cccfbd7dbd, 0x8038d51cb897789d},
+    {0xbf21e44003acdd2c, 0xe0470a63e6bd56c4},
+    {0xeeea5d5004981478, 0x1858ccfce06cac75},
+    {0x95527a5202df0ccb, 0x0f37801e0c43ebc9},
+    {0xbaa718e68396cffd, 0xd30560258f54e6bb},
+    {0xe950df20247c83fd, 0x47c6b82ef32a206a},
+    {0x91d28b7416cdd27e, 0x4cdc331d57fa5442},
+    {0xb6472e511c81471d, 0xe0133fe4adf8e953},
+    {0xe3d8f9e563a198e5, 0x58180fddd97723a7},
+    {0x8e679c2f5e44ff8f, 0x570f09eaa7ea7649},
+    {0xb201833b35d63f73, 0x2cd2cc6551e513db},
+    {0xde81e40a034bcf4f, 0xf8077f7ea65e58d2},
+    {0x8b112e86420f6191, 0xfb04afaf27faf783},
+    {0xadd57a27d29339f6, 0x79c5db9af1f9b564},
+    {0xd94ad8b1c7380874, 0x18375281ae7822bd},
+    {0x87cec76f1c830548, 0x8f2293910d0b15b6},
+    {0xa9c2794ae3a3c69a, 0xb2eb3875504ddb23},
+    {0xd433179d9c8cb841, 0x5fa60692a46151ec},
+    {0x849feec281d7f328, 0xdbc7c41ba6bcd334},
+    {0xa5c7ea73224deff3, 0x12b9b522906c0801},
+    {0xcf39e50feae16bef, 0xd768226b34870a01},
+    {0x81842f29f2cce375, 0xe6a1158300d46641},
+    {0xa1e53af46f801c53, 0x60495ae3c1097fd1},
+    {0xca5e89b18b602368, 0x385bb19cb14bdfc5},
+    {0xfcf62c1dee382c42, 0x46729e03dd9ed7b6},
+    {0x9e19db92b4e31ba9, 0x6c07a2c26a8346d2},
+    {0xc5a05277621be293, 0xc7098b7305241886},
+    {0xf70867153aa2db38, 0xb8cbee4fc66d1ea8}
+};

-struct cached_power // c = f * 2^e ~= 10^k
-{
-  std::uint64_t f;
-  int e;
-  int k;
+// Per-format helper routines (binary64 specializations of the Dragonbox steps).
+struct compute_mul_result {
+  std::uint64_t integer_part;
+  bool is_integer;
+};
+struct compute_mul_parity_result {
+  bool parity;
+  bool is_integer;
 };

-/*!
-For a normalized diyfp w = f * 2^e, this function returns a (normalized) cached
-power-of-ten c = f_c * 2^e_c, such that the exponent of the product w * c
-satisfies (Definition 3.2 from [1])
-     alpha <= e_c + e + q <= gamma.
-*/
-inline cached_power get_cached_power_for_binary_exponent(int e) {
-  // Now
-  //
-  //      alpha <= e_c + e + q <= gamma                                    (1)
-  //      ==> f_c * 2^alpha <= c * 2^e * 2^q
-  //
-  // and since the c's are normalized, 2^(q-1) <= f_c,
-  //
-  //      ==> 2^(q - 1 + alpha) <= c * 2^(e + q)
-  //      ==> 2^(alpha - e - 1) <= c
-  //
-  // If c were an exact power of ten, i.e. c = 10^k, one may determine k as
-  //
-  //      k = ceil( log_10( 2^(alpha - e - 1) ) )
-  //        = ceil( (alpha - e - 1) * log_10(2) )
-  //
-  // From the paper:
-  // "In theory the result of the procedure could be wrong since c is rounded,
-  //  and the computation itself is approximated [...]. In practice, however,
-  //  this simple function is sufficient."
-  //
-  // For IEEE double precision floating-point numbers converted into
-  // normalized diyfp's w = f * 2^e, with q = 64,
-  //
-  //      e >= -1022      (min IEEE exponent)
-  //           -52        (p - 1)
-  //           -52        (p - 1, possibly normalize denormal IEEE numbers)
-  //           -11        (normalize the diyfp)
-  //         = -1137
-  //
-  // and
-  //
-  //      e <= +1023      (max IEEE exponent)
-  //           -52        (p - 1)
-  //           -11        (normalize the diyfp)
-  //         = 960
-  //
-  // This binary exponent range [-1137,960] results in a decimal exponent
-  // range [-307,324]. One does not need to store a cached power for each
-  // k in this range. For each such k it suffices to find a cached power
-  // such that the exponent of the product lies in [alpha,gamma].
-  // This implies that the difference of the decimal exponents of adjacent
-  // table entries must be less than or equal to
-  //
-  //      floor( (gamma - alpha) * log_10(2) ) = 8.
-  //
-  // (A smaller distance gamma-alpha would require a larger table.)
-
-  // NB:
-  // Actually this function returns c, such that -60 <= e_c + e + 64 <= -34.
-
-  constexpr int kCachedPowersMinDecExp = -300;
-  constexpr int kCachedPowersDecStep = 8;
-
-  static constexpr std::array<cached_power, 79> kCachedPowers = {{
-      {0xAB70FE17C79AC6CA, -1060, -300}, {0xFF77B1FCBEBCDC4F, -1034, -292},
-      {0xBE5691EF416BD60C, -1007, -284}, {0x8DD01FAD907FFC3C, -980, -276},
-      {0xD3515C2831559A83, -954, -268},  {0x9D71AC8FADA6C9B5, -927, -260},
-      {0xEA9C227723EE8BCB, -901, -252},  {0xAECC49914078536D, -874, -244},
-      {0x823C12795DB6CE57, -847, -236},  {0xC21094364DFB5637, -821, -228},
-      {0x9096EA6F3848984F, -794, -220},  {0xD77485CB25823AC7, -768, -212},
-      {0xA086CFCD97BF97F4, -741, -204},  {0xEF340A98172AACE5, -715, -196},
-      {0xB23867FB2A35B28E, -688, -188},  {0x84C8D4DFD2C63F3B, -661, -180},
-      {0xC5DD44271AD3CDBA, -635, -172},  {0x936B9FCEBB25C996, -608, -164},
-      {0xDBAC6C247D62A584, -582, -156},  {0xA3AB66580D5FDAF6, -555, -148},
-      {0xF3E2F893DEC3F126, -529, -140},  {0xB5B5ADA8AAFF80B8, -502, -132},
-      {0x87625F056C7C4A8B, -475, -124},  {0xC9BCFF6034C13053, -449, -116},
-      {0x964E858C91BA2655, -422, -108},  {0xDFF9772470297EBD, -396, -100},
-      {0xA6DFBD9FB8E5B88F, -369, -92},   {0xF8A95FCF88747D94, -343, -84},
-      {0xB94470938FA89BCF, -316, -76},   {0x8A08F0F8BF0F156B, -289, -68},
-      {0xCDB02555653131B6, -263, -60},   {0x993FE2C6D07B7FAC, -236, -52},
-      {0xE45C10C42A2B3B06, -210, -44},   {0xAA242499697392D3, -183, -36},
-      {0xFD87B5F28300CA0E, -157, -28},   {0xBCE5086492111AEB, -130, -20},
-      {0x8CBCCC096F5088CC, -103, -12},   {0xD1B71758E219652C, -77, -4},
-      {0x9C40000000000000, -50, 4},      {0xE8D4A51000000000, -24, 12},
-      {0xAD78EBC5AC620000, 3, 20},       {0x813F3978F8940984, 30, 28},
-      {0xC097CE7BC90715B3, 56, 36},      {0x8F7E32CE7BEA5C70, 83, 44},
-      {0xD5D238A4ABE98068, 109, 52},     {0x9F4F2726179A2245, 136, 60},
-      {0xED63A231D4C4FB27, 162, 68},     {0xB0DE65388CC8ADA8, 189, 76},
-      {0x83C7088E1AAB65DB, 216, 84},     {0xC45D1DF942711D9A, 242, 92},
-      {0x924D692CA61BE758, 269, 100},    {0xDA01EE641A708DEA, 295, 108},
-      {0xA26DA3999AEF774A, 322, 116},    {0xF209787BB47D6B85, 348, 124},
-      {0xB454E4A179DD1877, 375, 132},    {0x865B86925B9BC5C2, 402, 140},
-      {0xC83553C5C8965D3D, 428, 148},    {0x952AB45CFA97A0B3, 455, 156},
-      {0xDE469FBD99A05FE3, 481, 164},    {0xA59BC234DB398C25, 508, 172},
-      {0xF6C69A72A3989F5C, 534, 180},    {0xB7DCBF5354E9BECE, 561, 188},
-      {0x88FCF317F22241E2, 588, 196},    {0xCC20CE9BD35C78A5, 614, 204},
-      {0x98165AF37B2153DF, 641, 212},    {0xE2A0B5DC971F303A, 667, 220},
-      {0xA8D9D1535CE3B396, 694, 228},    {0xFB9B7CD9A4A7443C, 720, 236},
-      {0xBB764C4CA7A44410, 747, 244},    {0x8BAB8EEFB6409C1A, 774, 252},
-      {0xD01FEF10A657842C, 800, 260},    {0x9B10A4E5E9913129, 827, 268},
-      {0xE7109BFBA19C0C9D, 853, 276},    {0xAC2820D9623BF429, 880, 284},
-      {0x80444B5E7AA7CF85, 907, 292},    {0xBF21E44003ACDD2D, 933, 300},
-      {0x8E679C2F5E44FF8F, 960, 308},    {0xD433179D9C8CB841, 986, 316},
-      {0x9E19DB92B4E31BA9, 1013, 324},
-  }};
-
-  // This computation gives exactly the same results for k as
-  //      k = ceil((kAlpha - e - 1) * 0.30102999566398114)
-  // for |e| <= 1500, but doesn't require floating-point operations.
-  // NB: log_10(2) ~= 78913 / 2^18
-  const int f = kAlpha - e - 1;
-  const int k = (f * 78913) / (1 << 18) + static_cast<int>(f > 0);
-
-  const int index = (-kCachedPowersMinDecExp + k + (kCachedPowersDecStep - 1)) /
-                    kCachedPowersDecStep;
-
-  const cached_power cached = kCachedPowers[static_cast<std::size_t>(index)];
-
-  return cached;
+inline compute_mul_result compute_mul(std::uint64_t u, uint128 c) noexcept {
+  const uint128 r = umul192_upper128(u, c);
+  return {r.high, r.low == 0};
 }

-/*!
-For n != 0, returns k, such that pow10 := 10^(k-1) <= n < 10^k.
-For n == 0, returns 1 and sets pow10 := 1.
-*/
-inline int find_largest_pow10(const std::uint32_t n, std::uint32_t &pow10) {
-  // LCOV_EXCL_START
-  if (n >= 1000000000) {
-    pow10 = 1000000000;
-    return 10;
-  }
-  // LCOV_EXCL_STOP
-  else if (n >= 100000000) {
-    pow10 = 100000000;
-    return 9;
-  } else if (n >= 10000000) {
-    pow10 = 10000000;
-    return 8;
-  } else if (n >= 1000000) {
-    pow10 = 1000000;
-    return 7;
-  } else if (n >= 100000) {
-    pow10 = 100000;
-    return 6;
-  } else if (n >= 10000) {
-    pow10 = 10000;
-    return 5;
-  } else if (n >= 1000) {
-    pow10 = 1000;
-    return 4;
-  } else if (n >= 100) {
-    pow10 = 100;
-    return 3;
-  } else if (n >= 10) {
-    pow10 = 10;
-    return 2;
-  } else {
-    pow10 = 1;
-    return 1;
-  }
+inline std::uint64_t compute_delta(uint128 c, int beta) noexcept {
+  return c.high >> (total_bits - 1 - beta);
 }

-inline void grisu2_round(char *buf, int len, std::uint64_t dist,
-                         std::uint64_t delta, std::uint64_t rest,
-                         std::uint64_t ten_k) {
-
-  //               <--------------------------- delta ---->
-  //                                  <---- dist --------->
-  // --------------[------------------+-------------------]--------------
-  //               M-                 w                   M+
-  //
-  //                                  ten_k
-  //                                <------>
-  //                                       <---- rest ---->
-  // --------------[------------------+----+--------------]--------------
-  //                                  w    V
-  //                                       = buf * 10^k
-  //
-  // ten_k represents a unit-in-the-last-place in the decimal representation
-  // stored in buf.
-  // Decrement buf by ten_k while this takes buf closer to w.
-
-  // The tests are written in this order to avoid overflow in unsigned
-  // integer arithmetic.
+inline compute_mul_parity_result compute_mul_parity(std::uint64_t two_f,
+                                                    uint128 c, int beta) noexcept {
+  // beta is always in [1, 63] here.
+  const uint128 r = umul192_lower128(two_f, c);
+  return {((r.high >> (64 - beta)) & 1) != 0,
+          ((r.high << beta) | (r.low >> (64 - beta))) == 0};
+}

-  while (rest < dist && delta - rest >= ten_k &&
-         (rest + ten_k < dist || dist - rest > rest + ten_k - dist)) {
-    buf[len - 1]--;
-    rest += ten_k;
-  }
+inline std::uint64_t
+compute_left_endpoint_for_shorter_interval_case(uint128 c, int beta) noexcept {
+  return (c.high - (c.high >> (significand_bits + 2))) >>
+         (total_bits - significand_bits - 1 - beta);
 }

-/*!
-Generates V = buffer * 10^decimal_exponent, such that M- <= V <= M+.
-M- and M+ must be normalized and share the same exponent -60 <= e <= -32.
-*/
-inline void grisu2_digit_gen(char *buffer, int &length, int &decimal_exponent,
-                             diyfp M_minus, diyfp w, diyfp M_plus) {
-  static_assert(kAlpha >= -60, "internal error");
-  static_assert(kGamma <= -32, "internal error");
+inline std::uint64_t
+compute_right_endpoint_for_shorter_interval_case(uint128 c, int beta) noexcept {
+  return (c.high + (c.high >> (significand_bits + 1))) >>
+         (total_bits - significand_bits - 1 - beta);
+}

-  // Generates the digits (and the exponent) of a decimal floating-point
-  // number V = buffer * 10^decimal_exponent in the range [M-, M+]. The diyfp's
-  // w, M- and M+ share the same exponent e, which satisfies alpha <= e <=
-  // gamma.
-  //
-  //               <--------------------------- delta ---->
-  //                                  <---- dist --------->
-  // --------------[------------------+-------------------]--------------
-  //               M-                 w                   M+
-  //
-  // Grisu2 generates the digits of M+ from left to right and stops as soon as
-  // V is in [M-,M+].
+inline std::uint64_t
+compute_round_up_for_shorter_interval_case(uint128 c, int beta) noexcept {
+  return ((c.high >> (total_bits - significand_bits - 2 - beta)) + 1) / 2;
+}

-  std::uint64_t delta =
-      diyfp::sub(M_plus, M_minus)
-          .f; // (significand of (M+ - M-), implicit exponent is e)
-  std::uint64_t dist =
-      diyfp::sub(M_plus, w)
-          .f; // (significand of (M+ - w ), implicit exponent is e)
+// floor(n / 10) for the shorter-interval right endpoint (n bounded so the
+// single multiply below is exact).
+inline std::uint64_t divide_by_pow10_1(std::uint64_t n) noexcept {
+  return umul128_upper64(n, std::uint64_t(1844674407370955162ull));
+}

-  // Split M+ = f * 2^e into two parts p1 and p2 (note: e < 0):
-  //
-  //      M+ = f * 2^e
-  //         = ((f div 2^-e) * 2^-e + (f mod 2^-e)) * 2^e
-  //         = ((p1        ) * 2^-e + (p2        )) * 2^e
-  //         = p1 + p2 * 2^e
+// floor(n / 1000) for the larger-divisor step (n bounded as above).
+inline std::uint64_t divide_by_pow10_3(std::uint64_t n) noexcept {
+  return umul128_upper64(n, std::uint64_t(4722366482869645214ull)) >> 8;
+}

-  const diyfp one(std::uint64_t{1} << -M_plus.e, M_plus.e);
+// Returns whether n is divisible by 10^kappa (= 100) and divides n by it.
+inline bool check_divisibility_and_divide_by_pow10_kappa(std::uint64_t &n) noexcept {
+  // magic number for division by 100 (kappa == 2).
+  const std::uint32_t prod = std::uint32_t(n) * std::uint32_t(656);
+  const bool result = (prod & 0xffffu) < 656u;
+  n = std::uint64_t(prod >> 16);
+  return result;
+}

-  auto p1 = static_cast<std::uint32_t>(
-      M_plus.f >>
-      -one.e); // p1 = f div 2^-e (Since -e >= 32, p1 fits into a 32-bit int.)
-  std::uint64_t p2 = M_plus.f & (one.f - 1); // p2 = f mod 2^-e
+// Strip trailing decimal zeros from significand, bumping exponent accordingly.
+// Branchless search; constants from the Dragonbox reference.
+inline void remove_trailing_zeros(std::uint64_t &significand, int &exponent) noexcept {
+  std::uint64_t r = rotr64(significand * std::uint64_t(28999941890838049ull), 8);
+  bool b = r < std::uint64_t(184467440738ull);
+  int s = b ? 1 : 0;
+  significand = b ? r : significand;

-  // 1)
-  //
-  // Generate the digits of the integral part p1 = d[n-1]...d[1]d[0]
+  r = rotr64(significand * std::uint64_t(182622766329724561ull), 4);
+  b = r < std::uint64_t(1844674407370956ull);
+  s = s * 2 + (b ? 1 : 0);
+  significand = b ? r : significand;

-  std::uint32_t pow10;
-  const int k = find_largest_pow10(p1, pow10);
+  r = rotr64(significand * std::uint64_t(10330176681277348905ull), 2);
+  b = r < std::uint64_t(184467440737095517ull);
+  s = s * 2 + (b ? 1 : 0);
+  significand = b ? r : significand;

-  //      10^(k-1) <= p1 < 10^k, pow10 = 10^(k-1)
-  //
-  //      p1 = (p1 div 10^(k-1)) * 10^(k-1) + (p1 mod 10^(k-1))
-  //         = (d[k-1]         ) * 10^(k-1) + (p1 mod 10^(k-1))
-  //
-  //      M+ = p1                                             + p2 * 2^e
-  //         = d[k-1] * 10^(k-1) + (p1 mod 10^(k-1))          + p2 * 2^e
-  //         = d[k-1] * 10^(k-1) + ((p1 mod 10^(k-1)) * 2^-e + p2) * 2^e
-  //         = d[k-1] * 10^(k-1) + (                         rest) * 2^e
-  //
-  // Now generate the digits d[n] of p1 from left to right (n = k-1,...,0)
-  //
-  //      p1 = d[k-1]...d[n] * 10^n + d[n-1]...d[0]
-  //
-  // but stop as soon as
-  //
-  //      rest * 2^e = (d[n-1]...d[0] * 2^-e + p2) * 2^e <= delta * 2^e
+  r = rotr64(significand * std::uint64_t(14757395258967641293ull), 1);
+  b = r < std::uint64_t(1844674407370955162ull);
+  s = s * 2 + (b ? 1 : 0);
+  significand = b ? r : significand;

-  int n = k;
-  while (n > 0) {
-    // Invariants:
-    //      M+ = buffer * 10^n + (p1 + p2 * 2^e)    (buffer = 0 for n = k)
-    //      pow10 = 10^(n-1) <= p1 < 10^n
-    //
-    const std::uint32_t d = p1 / pow10; // d = p1 div 10^(n-1)
-    const std::uint32_t r = p1 % pow10; // r = p1 mod 10^(n-1)
-    //
-    //      M+ = buffer * 10^n + (d * 10^(n-1) + r) + p2 * 2^e
-    //         = (buffer * 10 + d) * 10^(n-1) + (r + p2 * 2^e)
-    //
-    buffer[length++] = static_cast<char>('0' + d); // buffer := buffer * 10 + d
-    //
-    //      M+ = buffer * 10^(n-1) + (r + p2 * 2^e)
-    //
-    p1 = r;
-    n--;
-    //
-    //      M+ = buffer * 10^n + (p1 + p2 * 2^e)
-    //      pow10 = 10^n
-    //
+  exponent += s;
+}

-    // Now check if enough digits have been generated.
-    // Compute
-    //
-    //      p1 + p2 * 2^e = (p1 * 2^-e + p2) * 2^e = rest * 2^e
-    //
-    // Note:
-    // Since rest and delta share the same exponent e, it suffices to
-    // compare the significands.
-    const std::uint64_t rest = (std::uint64_t{p1} << -one.e) + p2;
-    if (rest <= delta) {
-      // V = buffer * 10^n, with M- <= V <= M+.
+// Dragonbox core: shortest (significand, exponent) such that
+//   value == significand * 10^exponent
+// for a finite, positive, non-zero binary64 value, decomposed into its raw
+// significand bits and biased exponent bits.
+struct decimal_fp {
+  std::uint64_t significand;
+  int exponent;
+};

-      decimal_exponent += n;
+inline decimal_fp to_decimal(std::uint64_t binary_significand,
+                             int binary_exponent) noexcept {
+  const bool is_even = (binary_significand % 2 == 0);
+  std::uint64_t two_fc = binary_significand * 2;
+
+  // Is the input a normal number?
+  if (binary_exponent != 0) {
+    binary_exponent += exponent_bias - significand_bits;
+
+    // Shorter interval case; proceed like Schubfach.
+    if (two_fc == 0) {
+      const int minus_k =
+          floor_log10_pow2_minus_log10_4_over_3(binary_exponent);
+      const int beta = binary_exponent + floor_log2_pow10(-minus_k);
+      const uint128 c = cache[-minus_k - cache_min_k];
+
+      std::uint64_t xi =
+          compute_left_endpoint_for_shorter_interval_case(c, beta);
+      const std::uint64_t zi =
+          compute_right_endpoint_for_shorter_interval_case(c, beta);
+
+      // If the left endpoint is not an integer, increase it. (Both endpoints
+      // are always included since the significand is even.)
+      if (!(binary_exponent >=
+                case_shorter_interval_left_endpoint_lower_threshold &&
+            binary_exponent <=
+                case_shorter_interval_left_endpoint_upper_threshold)) {
+        ++xi;
+      }

-      // We may now just stop. But instead look if the buffer could be
-      // decremented to bring V closer to w.
-      //
-      // pow10 = 10^n is now 1 ulp in the decimal representation V.
-      // The rounding procedure works with diyfp's with an implicit
-      // exponent of e.
-      //
-      //      10^n = (10^n * 2^-e) * 2^e = ulp * 2^e
-      //
-      const std::uint64_t ten_n = std::uint64_t{pow10} << -one.e;
-      grisu2_round(buffer, length, dist, delta, rest, ten_n);
+      // Try the bigger divisor.
+      std::uint64_t decimal_significand = divide_by_pow10_1(zi);
+      if (decimal_significand * 10 >= xi) {
+        int decimal_exponent = minus_k + 1;
+        remove_trailing_zeros(decimal_significand, decimal_exponent);
+        return {decimal_significand, decimal_exponent};
+      }

-      return;
+      // Otherwise, compute the round-up of y.
+      decimal_significand =
+          compute_round_up_for_shorter_interval_case(c, beta);
+      // On a tie, choose the even one.
+      if ((decimal_significand % 2 != 0) &&
+          binary_exponent >= shorter_interval_tie_lower_threshold &&
+          binary_exponent <= shorter_interval_tie_upper_threshold) {
+        --decimal_significand;
+      } else if (decimal_significand < xi) {
+        ++decimal_significand;
+      }
+      return {decimal_significand, minus_k};
     }

-    pow10 /= 10;
-    //
-    //      pow10 = 10^(n-1) <= p1 < 10^n
-    // Invariants restored.
-  }
-
-  // 2)
-  //
-  // The digits of the integral part have been generated:
-  //
-  //      M+ = d[k-1]...d[1]d[0] + p2 * 2^e
-  //         = buffer            + p2 * 2^e
-  //
-  // Now generate the digits of the fractional part p2 * 2^e.
-  //
-  // Note:
-  // No decimal point is generated: the exponent is adjusted instead.
-  //
-  // p2 actually represents the fraction
-  //
-  //      p2 * 2^e
-  //          = p2 / 2^-e
-  //          = d[-1] / 10^1 + d[-2] / 10^2 + ...
-  //
-  // Now generate the digits d[-m] of p1 from left to right (m = 1,2,...)
-  //
-  //      p2 * 2^e = d[-1]d[-2]...d[-m] * 10^-m
-  //                      + 10^-m * (d[-m-1] / 10^1 + d[-m-2] / 10^2 + ...)
-  //
-  // using
-  //
-  //      10^m * p2 = ((10^m * p2) div 2^-e) * 2^-e + ((10^m * p2) mod 2^-e)
-  //                = (                   d) * 2^-e + (                   r)
-  //
-  // or
-  //      10^m * p2 * 2^e = d + r * 2^e
-  //
-  // i.e.
-  //
-  //      M+ = buffer + p2 * 2^e
-  //         = buffer + 10^-m * (d + r * 2^e)
-  //         = (buffer * 10^m + d) * 10^-m + 10^-m * r * 2^e
-  //
-  // and stop as soon as 10^-m * r * 2^e <= delta * 2^e
-
-  int m = 0;
-  for (;;) {
-    // Invariant:
-    //      M+ = buffer * 10^-m + 10^-m * (d[-m-1] / 10 + d[-m-2] / 10^2 + ...)
-    //      * 2^e
-    //         = buffer * 10^-m + 10^-m * (p2                                 )
-    //         * 2^e = buffer * 10^-m + 10^-m * (1/10 * (10 * p2) ) * 2^e =
-    //         buffer * 10^-m + 10^-m * (1/10 * ((10*p2 div 2^-e) * 2^-e +
-    //         (10*p2 mod 2^-e)) * 2^e
-    //
-    p2 *= 10;
-    const std::uint64_t d = p2 >> -one.e;     // d = (10 * p2) div 2^-e
-    const std::uint64_t r = p2 & (one.f - 1); // r = (10 * p2) mod 2^-e
-    //
-    //      M+ = buffer * 10^-m + 10^-m * (1/10 * (d * 2^-e + r) * 2^e
-    //         = buffer * 10^-m + 10^-m * (1/10 * (d + r * 2^e))
-    //         = (buffer * 10 + d) * 10^(-m-1) + 10^(-m-1) * r * 2^e
-    //
-    buffer[length++] = static_cast<char>('0' + d); // buffer := buffer * 10 + d
-    //
-    //      M+ = buffer * 10^(-m-1) + 10^(-m-1) * r * 2^e
-    //
-    p2 = r;
-    m++;
-    //
-    //      M+ = buffer * 10^-m + 10^-m * p2 * 2^e
-    // Invariant restored.
-
-    // Check if enough digits have been generated.
-    //
-    //      10^-m * p2 * 2^e <= delta * 2^e
-    //              p2 * 2^e <= 10^m * delta * 2^e
-    //                    p2 <= 10^m * delta
-    delta *= 10;
-    dist *= 10;
-    if (p2 <= delta) {
+    // Normal interval case.
+    two_fc |= (std::uint64_t(1) << (significand_bits + 1));
+  } else {
+    // Subnormal number: normal interval case.
+    binary_exponent = min_exponent - significand_bits;
+  }
+
+  // Step 1: Schubfach multiplier calculation.
+  const int minus_k = floor_log10_pow2(binary_exponent) - kappa;
+  const uint128 c = cache[-minus_k - cache_min_k];
+  const int beta = binary_exponent + floor_log2_pow10(-minus_k);
+
+  const std::uint64_t deltai = compute_delta(c, beta);
+  const compute_mul_result z_result =
+      compute_mul((two_fc | 1) << beta, c);
+
+  // Step 2: Try larger divisor; remove trailing zeros if necessary.
+  std::uint64_t decimal_significand = divide_by_pow10_3(z_result.integer_part);
+  std::uint64_t r =
+      z_result.integer_part - std::uint64_t(big_divisor) * decimal_significand;
+
+  do {
+    if (r < deltai) {
+      // Exclude the right endpoint if necessary.
+      if ((r | std::uint64_t(!z_result.is_integer) | std::uint64_t(is_even)) ==
+          0) {
+        --decimal_significand;
+        r = big_divisor;
+        break;
+      }
+    } else if (r > deltai) {
       break;
+    } else {
+      // r == deltai; compare fractional parts.
+      const compute_mul_parity_result x_result =
+          compute_mul_parity(two_fc - 1, c, beta);
+      if (!(x_result.parity | (x_result.is_integer & is_even))) {
+        break;
+      }
     }
-  }
-
-  // V = buffer * 10^-m, with M- <= V <= M+.
-
-  decimal_exponent -= m;
-
-  // 1 ulp in the decimal representation is now 10^-m.
-  // Since delta and dist are now scaled by 10^m, we need to do the
-  // same with ulp in order to keep the units in sync.
-  //
-  //      10^m * 10^-m = 1 = 2^-e * 2^e = ten_m * 2^e
-  //
-  const std::uint64_t ten_m = one.f;
-  grisu2_round(buffer, length, dist, delta, p2, ten_m);

-  // By construction this algorithm generates the shortest possible decimal
-  // number (Loitsch, Theorem 6.2) which rounds back to w.
-  // For an input number of precision p, at least
-  //
-  //      N = 1 + ceil(p * log_10(2))
-  //
-  // decimal digits are sufficient to identify all binary floating-point
-  // numbers (Matula, "In-and-Out conversions").
-  // This implies that the algorithm does not produce more than N decimal
-  // digits.
-  //
-  //      N = 17 for p = 53 (IEEE double precision)
-  //      N = 9  for p = 24 (IEEE single precision)
-}
+    int decimal_exponent = minus_k + kappa + 1;
+    remove_trailing_zeros(decimal_significand, decimal_exponent);
+    return {decimal_significand, decimal_exponent};
+  } while (false);

-/*!
-v = buf * 10^decimal_exponent
-len is the length of the buffer (number of decimal digits)
-The buffer must be large enough, i.e. >= max_digits10.
-*/
-inline void grisu2(char *buf, int &len, int &decimal_exponent, diyfp m_minus,
-                   diyfp v, diyfp m_plus) {
-
-  //  --------(-----------------------+-----------------------)--------    (A)
-  //          m-                      v                       m+
-  //
-  //  --------------------(-----------+-----------------------)--------    (B)
-  //                      m-          v                       m+
-  //
-  // First scale v (and m- and m+) such that the exponent is in the range
-  // [alpha, gamma].
+  // Step 3: Find the significand with the smaller divisor.
+  decimal_significand *= 10;

-  const cached_power cached = get_cached_power_for_binary_exponent(m_plus.e);
+  std::uint64_t dist = r - (deltai / 2) + (small_divisor / 2);
+  const bool approx_y_parity = ((dist ^ (small_divisor / 2)) & 1) != 0;

-  const diyfp c_minus_k(cached.f, cached.e); // = c ~= 10^-k
+  const bool divisible_by_small_divisor =
+      check_divisibility_and_divide_by_pow10_kappa(dist);

-  // The exponent of the products is = v.e + c_minus_k.e + q and is in the range
-  // [alpha,gamma]
-  const diyfp w = diyfp::mul(v, c_minus_k);
-  const diyfp w_minus = diyfp::mul(m_minus, c_minus_k);
-  const diyfp w_plus = diyfp::mul(m_plus, c_minus_k);
+  decimal_significand += dist;

-  //  ----(---+---)---------------(---+---)---------------(---+---)----
-  //          w-                      w                       w+
-  //          = c*m-                  = c*v                   = c*m+
-  //
-  // diyfp::mul rounds its result and c_minus_k is approximated too. w, w- and
-  // w+ are now off by a small amount.
-  // In fact:
-  //
-  //      w - v * 10^k < 1 ulp
-  //
-  // To account for this inaccuracy, add resp. subtract 1 ulp.
-  //
-  //  --------+---[---------------(---+---)---------------]---+--------
-  //          w-  M-                  w                   M+  w+
-  //
-  // Now any number in [M-, M+] (bounds included) will round to w when input,
-  // regardless of how the input rounding algorithm breaks ties.
-  //
-  // And digit_gen generates the shortest possible such number in [M-, M+].
-  // Note that this does not mean that Grisu2 always generates the shortest
-  // possible number in the interval (m-, m+).
-  const diyfp M_minus(w_minus.f + 1, w_minus.e);
-  const diyfp M_plus(w_plus.f - 1, w_plus.e);
-
-  decimal_exponent = -cached.k; // = -(-k) = k
+  if (divisible_by_small_divisor) {
+    const compute_mul_parity_result y_result =
+        compute_mul_parity(two_fc, c, beta);
+    if (y_result.parity != approx_y_parity) {
+      --decimal_significand;
+    } else if ((decimal_significand % 2) != 0 && y_result.is_integer) {
+      // On a tie (y is an integer), choose the even one.
+      --decimal_significand;
+    }
+  }

-  grisu2_digit_gen(buf, len, decimal_exponent, M_minus, w, M_plus);
+  return {decimal_significand, minus_k + kappa};
 }

 /*!
-v = buf * 10^decimal_exponent
-len is the length of the buffer (number of decimal digits)
-The buffer must be large enough, i.e. >= max_digits10.
+Fills buf with the shortest decimal digits of 'value' (which must be finite,
+positive and non-zero), sets len to the number of digits, and decimal_exponent
+so that value == (buf, interpreted as an integer) * 10^decimal_exponent.
+This mirrors the contract of the previous grisu2() entry point.
 */
-template <typename FloatType>
-void grisu2(char *buf, int &len, int &decimal_exponent, FloatType value) {
-  static_assert(diyfp::kPrecision >= std::numeric_limits<FloatType>::digits + 3,
-                "internal error: not enough precision");
-
-  // If the neighbors (and boundaries) of 'value' are always computed for
-  // double-precision numbers, all float's can be recovered using strtod (and
-  // strtof). However, the resulting decimal representations are not exactly
-  // "short".
-  //
-  // The documentation for 'std::to_chars'
-  // (https://en.cppreference.com/w/cpp/utility/to_chars) says "value is
-  // converted to a string as if by std::sprintf in the default ("C") locale"
-  // and since sprintf promotes float's to double's, I think this is exactly
-  // what 'std::to_chars' does. On the other hand, the documentation for
-  // 'std::to_chars' requires that "parsing the representation using the
-  // corresponding std::from_chars function recovers value exactly". That
-  // indicates that single precision floating-point numbers should be recovered
-  // using 'std::strtof'.
-  //
-  // NB: If the neighbors are computed for single-precision numbers, there is a
-  // single float
-  //     (7.0385307e-26f) which can't be recovered using strtod. The resulting
-  //     double precision value is off by 1 ulp.
-#if 0
-    const boundaries w = compute_boundaries(static_cast<double>(value));
-#else
-  const boundaries w = compute_boundaries(value);
-#endif
-
-  grisu2(buf, len, decimal_exponent, w.minus, w.w, w.plus);
+inline void dragonbox(char *buf, int &len, int &decimal_exponent,
+                      double value) {
+  std::uint64_t bits;
+  std::memcpy(&bits, &value, sizeof(bits));
+  const std::uint64_t binary_significand =
+      bits & ((std::uint64_t(1) << significand_bits) - 1);
+  const int binary_exponent =
+      int((bits >> significand_bits) & 0x7ff);
+
+  const decimal_fp dec = to_decimal(binary_significand, binary_exponent);
+
+  // Convert the decimal significand to digits:
+  // 1) Proceed 2 digits at a time (s % 100) via a 00..99 lookup table
+  //     (see Alexandrescu, "Three Optimization Tips for C++", 2012),
+  // 2) Digits come out least-significant first, writing them back-to-front
+  //     with p = tmp + sizeof(tmp); to avoid reversal pass
+  // 3) Proceed remaining digits after loop to avoid branchs inside it
+  // 4) memcpy digits to char* buf (inside function input)
+  static const char digits2[201] =
+      "0001020304050607080910111213141516171819"
+      "2021222324252627282930313233343536373839"
+      "4041424344454647484950515253545556575859"
+      "6061626364656667686970717273747576777879"
+      "8081828384858687888990919293949596979899";
+  // Digit area is 24; +16 padding lets us always memcpy 16 (+1) bytes with a
+  // compile-time size so the compiler inlines (no libc size-class branches).
+  // Callers must provide to_chars_buffer_size (40) bytes for the same reason.
+  constexpr int digit_area = 24;
+  char tmp[digit_area + 16];
+  char *p = tmp + digit_area; // write backward
+  std::uint64_t s = dec.significand;
+  while (s >= 100) {
+    const std::uint32_t idx = static_cast<std::uint32_t>(s % 100) * 2;
+    s /= 100;
+    p -= 2;
+    p[0] = digits2[idx];
+    p[1] = digits2[idx + 1];
+  }
+  if (s >= 10) {
+    const std::uint32_t idx = static_cast<std::uint32_t>(s) * 2;
+    p -= 2;
+    p[0] = digits2[idx];
+    p[1] = digits2[idx + 1];
+  } else {
+    *--p = static_cast<char>('0' + s);
+  }
+  // Fixed-size copy: double has at most 17 significant digits.
+  std::memcpy(buf, p, 16);
+  buf[16] = p[16];
+  len = static_cast<int>(tmp + digit_area - p);
+  decimal_exponent = dec.exponent;
 }

 /*!
@@ -4111,11 +4625,16 @@ inline char *format_buffer(char *buf, int len, int decimal_exponent,
   // k is the length of the buffer (number of decimal digits)
   // n is the position of the decimal point relative to the start of the buffer.

+  // All mem* sizes below are compile-time constants so the compiler inlines
+  // them as plain loads/stores. That requires over-writing past the logical
+  // string length; callers must reserve to_chars_buffer_size (40) bytes.
+  // Logical output is still bounded by ~24 characters; only the returned
+  // pointer reflects the true length.
+
   if (k <= n && n <= max_exp) {
     // digits[000]
     // len <= max_exp + 2
-
-    std::memset(buf + k, '0', static_cast<size_t>(n) - static_cast<size_t>(k));
+    std::memset(buf + k, '0', 16);
     // Make it look like a floating-point number (#362, #378)
     buf[n + 0] = '.';
     buf[n + 1] = '0';
@@ -4125,690 +4644,5307 @@ inline char *format_buffer(char *buf, int len, int decimal_exponent,
   if (0 < n && n <= max_exp) {
     // dig.its
     // len <= max_digits10 + 1
-    std::memmove(buf + (static_cast<size_t>(n) + 1), buf + n,
-                 static_cast<size_t>(k) - static_cast<size_t>(n));
+    // Shift the fractional digits one place right via a temp (overlap).
+    char shifted[16];
+    std::memcpy(shifted, buf + static_cast<size_t>(n), 16);
+    std::memcpy(buf + (static_cast<size_t>(n) + 1), shifted, 16);
     buf[n] = '.';
     return buf + (static_cast<size_t>(k) + 1U);
   }

-  if (min_exp < n && n <= 0) {
-    // 0.[000]digits
-    // len <= 2 + (-min_exp - 1) + max_digits10
+  if (min_exp < n && n <= 0) {
+    // 0.[000]digits
+    // With kMinExp = -4, n is in {-3,-2,-1,0}, so pad = -n is 0..3.
+    // len <= 2 + (-min_exp - 1) + max_digits10
+    char digits[17];
+    std::memcpy(digits, buf, 17);
+    const size_t pad = static_cast<size_t>(-n); // 0..3
+    buf[0] = '0';
+    buf[1] = '.';
+    // Fixed upper bound on leading zeros; only the first `pad` matter.
+    std::memset(buf + 2, '0', 4);
+    std::memcpy(buf + 2 + pad, digits, 17);
+    return buf + (2U + pad + static_cast<size_t>(k));
+  }
+
+  if (k == 1) {
+    // dE+123
+    // len <= 1 + 5
+    buf += 1;
+  } else {
+    // d.igitsE+123
+    // len <= max_digits10 + 1 + 5
+    // k-1 <= 16 for double; fixed-size shift via temp (overlap).
+    char shifted[16];
+    std::memcpy(shifted, buf + 1, 16);
+    std::memcpy(buf + 2, shifted, 16);
+    buf[1] = '.';
+    buf += 1 + static_cast<size_t>(k);
+  }
+
+  *buf++ = 'e';
+  return append_exponent(buf, n - 1);
+}
+
+} // NS dtoa_impl
+
+/*!
+The format of the resulting decimal representation is similar to printf's %g
+format. Returns an iterator pointing past-the-end of the decimal representation.
+@note The input number must be finite, i.e. NaN's and Inf's are not supported.
+@note The buffer must have at least to_chars_buffer_size (40) writable bytes.
+  Only ~24 characters are ever part of the logical result, but fixed-size
+  16/17-byte mem* over-writes require the extra scratch for safety.
+@note The result is NOT null-terminated.
+*/
+char *to_chars(char *first, const char *last, double value) {
+  static_cast<void>(last); // maybe unused - fix warning
+  bool negative = std::signbit(value);
+  if (negative) {
+    value = -value;
+    *first++ = '-';
+  }
+
+  if (value == 0) // +-0
+  {
+    *first++ = '0';
+    // Make it look like a floating-point number (#362, #378)
+    *first++ = '.';
+    *first++ = '0';
+    return first;
+  }
+  // Compute v = buffer * 10^decimal_exponent.
+  // The decimal digits are stored in the buffer, which needs to be interpreted
+  // as an unsigned decimal integer.
+  // len is the length of the buffer, i.e. the number of decimal digits.
+  int len = 0;
+  int decimal_exponent = 0;
+  dtoa_impl::dragonbox(first, len, decimal_exponent, value);
+  // Format the buffer like printf("%.*g", prec, value)
+  constexpr int kMinExp = -4;
+  constexpr int kMaxExp = std::numeric_limits<double>::digits10;
+
+  return dtoa_impl::format_buffer(first, len, decimal_exponent, kMinExp,
+                                  kMaxExp);
+}
+} // NS internal
+} // NS simdjson
+
+#endif // SIMDJSON_SRC_TO_CHARS_CPP
+
+
+/* end file to_chars.cpp */
+/* including from_chars.cpp: #include <from_chars.cpp> */
+/* begin file from_chars.cpp */
+#ifndef SIMDJSON_SRC_FROM_CHARS_CPP
+#define SIMDJSON_SRC_FROM_CHARS_CPP
+
+/* skipped duplicate #include <base.h> */
+
+/* including simdjson/internal/fast_float.h: #include "simdjson/internal/fast_float.h" */
+/* begin file simdjson/internal/fast_float.h */
+// Vendored from fast_float v8.2.10, generated by tools/vendor_fast_float.sh.
+// Do not edit by hand; re-run the script to update.
+//
+//   https://github.com/fastfloat/fast_float
+//   Licensed under Apache-2.0 OR MIT OR BSL-1.0, at your option.
+//
+// simdjson uses this for two things that its own number parser cannot do:
+//   * the slow path for numbers with more than 19 significant digits, where
+//     fast_float's bigint comparison is several times quicker than the
+//     Wuffs-derived decimal shifting it replaced (see src/from_chars.cpp), and
+//   * correctly rounded parsing inside a constant expression, which the runtime
+//     path cannot offer because it relies on memcpy and __uint128_t (see
+//     compile_time_json-inl.h).
+//
+// Two edits are applied by the script. Every fast_float name is rewritten so
+// that this copy cannot collide with a copy of fast_float that the surrounding
+// program includes for itself: namespace fast_float -> simdjson_fast_float,
+// FASTFLOAT_* -> SIMDJSON_FASTFLOAT_*, fastfloat_* -> simdjson_fastfloat_*. And
+// the accented letters in the attribution comments below are folded to ASCII,
+// to keep the tree ASCII-only; no disrespect to the people named is intended.
+// simdjson_fast_float by Daniel Lemire
+// simdjson_fast_float by Joao Paulo Magalhaes
+//
+//
+// with contributions from Eugene Golushkov
+// with contributions from Maksim Kita
+// with contributions from Marcin Wojdyr
+// with contributions from Neal Richardson
+// with contributions from Tim Paine
+// with contributions from Fabio Pellacini
+// with contributions from Lenard Szolnoki
+// with contributions from Jan Pharago
+// with contributions from Maya Warrier
+// with contributions from Taha Khokhar
+// with contributions from Anders Dalvander
+//
+//
+// Licensed under the Apache License, Version 2.0, or the
+// MIT License or the Boost License. This file may not be copied,
+// modified, or distributed except according to those terms.
+//
+// MIT License Notice
+//
+//    MIT License
+//
+//    Copyright (c) 2021 The simdjson_fast_float authors
+//
+//    Permission is hereby granted, free of charge, to any
+//    person obtaining a copy of this software and associated
+//    documentation files (the "Software"), to deal in the
+//    Software without restriction, including without
+//    limitation the rights to use, copy, modify, merge,
+//    publish, distribute, sublicense, and/or sell copies of
+//    the Software, and to permit persons to whom the Software
+//    is furnished to do so, subject to the following
+//    conditions:
+//
+//    The above copyright notice and this permission notice
+//    shall be included in all copies or substantial portions
+//    of the Software.
+//
+//    THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF
+//    ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED
+//    TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A
+//    PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT
+//    SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY
+//    CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION
+//    OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR
+//    IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
+//    DEALINGS IN THE SOFTWARE.
+//
+// Apache License (Version 2.0) Notice
+//
+//    Copyright 2021 The simdjson_fast_float authors
+//    Licensed under the Apache License, Version 2.0 (the "License");
+//    you may not use this file except in compliance with the License.
+//    You may obtain a copy of the License at
+//
+//    http://www.apache.org/licenses/LICENSE-2.0
+//
+//    Unless required by applicable law or agreed to in writing, software
+//    distributed under the License is distributed on an "AS IS" BASIS,
+//    WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+//    See the License for the specific language governing permissions and
+//
+// BOOST License Notice
+//
+//    Boost Software License - Version 1.0 - August 17th, 2003
+//
+//    Permission is hereby granted, free of charge, to any person or organization
+//    obtaining a copy of the software and accompanying documentation covered by
+//    this license (the "Software") to use, reproduce, display, distribute,
+//    execute, and transmit the Software, and to prepare derivative works of the
+//    Software, and to permit third-parties to whom the Software is furnished to
+//    do so, all subject to the following:
+//
+//    The copyright notices in the Software and this entire statement, including
+//    the above license grant, this restriction and the following disclaimer,
+//    must be included in all copies of the Software, in whole or in part, and
+//    all derivative works of the Software, unless such copies or derivative
+//    works are solely in the form of machine-executable object code generated by
+//    a source language processor.
+//
+//    THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
+//    IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
+//    FITNESS FOR A PARTICULAR PURPOSE, TITLE AND NON-INFRINGEMENT. IN NO EVENT
+//    SHALL THE COPYRIGHT HOLDERS OR ANYONE DISTRIBUTING THE SOFTWARE BE LIABLE
+//    FOR ANY DAMAGES OR OTHER LIABILITY, WHETHER IN CONTRACT, TORT OR OTHERWISE,
+//    ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
+//    DEALINGS IN THE SOFTWARE.
+//
+
+#ifndef SIMDJSON_FASTFLOAT_CONSTEXPR_FEATURE_DETECT_H
+#define SIMDJSON_FASTFLOAT_CONSTEXPR_FEATURE_DETECT_H
+
+#ifdef __has_include
+#if __has_include(<version>)
+#include <version>
+#endif
+#endif
+
+// Testing for https://wg21.link/N3652, adopted in C++14
+#if defined(__cpp_constexpr) && __cpp_constexpr >= 201304
+#define SIMDJSON_FASTFLOAT_CONSTEXPR14 constexpr
+#else
+#define SIMDJSON_FASTFLOAT_CONSTEXPR14
+#endif
+
+#if defined(__cpp_lib_bit_cast) && __cpp_lib_bit_cast >= 201806L
+#define SIMDJSON_FASTFLOAT_HAS_BIT_CAST 1
+#else
+#define SIMDJSON_FASTFLOAT_HAS_BIT_CAST 0
+#endif
+
+#if defined(__cpp_lib_is_constant_evaluated) &&                                \
+    __cpp_lib_is_constant_evaluated >= 201811L
+#define SIMDJSON_FASTFLOAT_HAS_IS_CONSTANT_EVALUATED 1
+#else
+#define SIMDJSON_FASTFLOAT_HAS_IS_CONSTANT_EVALUATED 0
+#endif
+
+#if defined(__cpp_if_constexpr) && __cpp_if_constexpr >= 201606L
+#define SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(x) if constexpr (x)
+#else
+#define SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(x) if (x)
+#endif
+
+// Testing for relevant C++20 constexpr library features
+#if SIMDJSON_FASTFLOAT_HAS_IS_CONSTANT_EVALUATED && SIMDJSON_FASTFLOAT_HAS_BIT_CAST &&           \
+    defined(__cpp_lib_constexpr_algorithms) &&                                 \
+    __cpp_lib_constexpr_algorithms >= 201806L /*For std::copy and std::fill*/
+#define SIMDJSON_FASTFLOAT_CONSTEXPR20 constexpr
+#define SIMDJSON_FASTFLOAT_IS_CONSTEXPR 1
+#else
+#define SIMDJSON_FASTFLOAT_CONSTEXPR20
+#define SIMDJSON_FASTFLOAT_IS_CONSTEXPR 0
+#endif
+
+#if __cplusplus >= 201703L || (defined(_MSVC_LANG) && _MSVC_LANG >= 201703L)
+#define SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE 0
+#else
+#define SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE 1
+#endif
+
+#endif // SIMDJSON_FASTFLOAT_CONSTEXPR_FEATURE_DETECT_H
+
+#ifndef SIMDJSON_FASTFLOAT_FLOAT_COMMON_H
+#define SIMDJSON_FASTFLOAT_FLOAT_COMMON_H
+
+#include <cfloat>
+#include <cstddef>
+#include <cstdint>
+#include <cassert>
+#include <cstring>
+#include <limits>
+#include <type_traits>
+#include <system_error>
+#ifdef __has_include
+#if __has_include(<stdfloat>) && (__cplusplus > 202002L || (defined(_MSVC_LANG) && (_MSVC_LANG > 202002L)))
+#include <stdfloat>
+#endif
+#endif
+
+#define SIMDJSON_FASTFLOAT_VERSION_MAJOR 8
+#define SIMDJSON_FASTFLOAT_VERSION_MINOR 2
+#define SIMDJSON_FASTFLOAT_VERSION_PATCH 10
+
+#define SIMDJSON_FASTFLOAT_STRINGIZE_IMPL(x) #x
+#define SIMDJSON_FASTFLOAT_STRINGIZE(x) SIMDJSON_FASTFLOAT_STRINGIZE_IMPL(x)
+
+#define SIMDJSON_FASTFLOAT_VERSION_STR                                                  \
+  SIMDJSON_FASTFLOAT_STRINGIZE(SIMDJSON_FASTFLOAT_VERSION_MAJOR)                                 \
+  "." SIMDJSON_FASTFLOAT_STRINGIZE(SIMDJSON_FASTFLOAT_VERSION_MINOR) "." SIMDJSON_FASTFLOAT_STRINGIZE(    \
+      SIMDJSON_FASTFLOAT_VERSION_PATCH)
+
+#define SIMDJSON_FASTFLOAT_VERSION                                                      \
+  (SIMDJSON_FASTFLOAT_VERSION_MAJOR * 10000 + SIMDJSON_FASTFLOAT_VERSION_MINOR * 100 +           \
+   SIMDJSON_FASTFLOAT_VERSION_PATCH)
+
+namespace simdjson_fast_float {
+
+enum class chars_format : uint64_t;
+
+namespace detail {
+constexpr chars_format basic_json_fmt = chars_format(1 << 5);
+constexpr chars_format basic_fortran_fmt = chars_format(1 << 6);
+} // namespace detail
+
+enum class chars_format : uint64_t {
+  scientific = 1 << 0,
+  fixed = 1 << 2,
+  hex = 1 << 3,
+  no_infnan = 1 << 4,
+  // RFC 8259: https://datatracker.ietf.org/doc/html/rfc8259#section-6
+  json = uint64_t(detail::basic_json_fmt) | fixed | scientific | no_infnan,
+  // Extension of RFC 8259 where, e.g., "inf" and "nan" are allowed.
+  json_or_infnan = uint64_t(detail::basic_json_fmt) | fixed | scientific,
+  fortran = uint64_t(detail::basic_fortran_fmt) | fixed | scientific,
+  general = fixed | scientific,
+  allow_leading_plus = 1 << 7,
+  skip_white_space = 1 << 8,
+};
+
+template <typename UC> struct from_chars_result_t {
+  UC const *ptr;
+  std::errc ec;
+
+  // https://www.open-std.org/jtc1/sc22/wg21/docs/papers/2023/p2497r0.html
+  constexpr explicit operator bool() const noexcept {
+    return ec == std::errc();
+  }
+};
+
+using from_chars_result = from_chars_result_t<char>;
+
+template <typename UC> struct parse_options_t {
+  constexpr explicit parse_options_t(chars_format fmt = chars_format::general,
+                                     UC dot = UC('.'), int b = 10)
+      : format(fmt), decimal_point(dot), base(b) {}
+
+  /** Which number formats are accepted */
+  chars_format format;
+  /** The character used as decimal point */
+  UC decimal_point;
+  /** The base used for integers */
+  int base;
+};
+
+using parse_options = parse_options_t<char>;
+
+} // namespace simdjson_fast_float
+
+#if SIMDJSON_FASTFLOAT_HAS_BIT_CAST
+#include <bit>
+#endif
+
+#if (defined(__x86_64) || defined(__x86_64__) || defined(_M_X64) ||            \
+     defined(__amd64) || defined(__aarch64__) || defined(_M_ARM64) ||          \
+     defined(__MINGW64__) || defined(__s390x__) ||                             \
+     (defined(__ppc64__) || defined(__PPC64__) || defined(__ppc64le__) ||      \
+      defined(__PPC64LE__)) ||                                                 \
+     defined(__loongarch64) || (defined(__riscv) && __riscv_xlen == 64))
+#define SIMDJSON_FASTFLOAT_64BIT 1
+#elif (defined(__i386) || defined(__i386__) || defined(_M_IX86) ||             \
+       defined(__arm__) || defined(_M_ARM) || defined(__ppc__) ||              \
+       defined(__MINGW32__) || defined(__EMSCRIPTEN__) ||                      \
+       (defined(__riscv) && __riscv_xlen == 32))
+#define SIMDJSON_FASTFLOAT_32BIT 1
+#else
+  // Need to check incrementally, since SIZE_MAX is a size_t, avoid overflow.
+// We can never tell the register width, but the SIZE_MAX is a good
+// approximation. UINTPTR_MAX and INTPTR_MAX are optional, so avoid them for max
+// portability.
+#if SIZE_MAX == 0xffff
+#error Unknown platform (16-bit, unsupported)
+#elif SIZE_MAX == 0xffffffff
+#define SIMDJSON_FASTFLOAT_32BIT 1
+#elif SIZE_MAX == 0xffffffffffffffff
+#define SIMDJSON_FASTFLOAT_64BIT 1
+#else
+#error Unknown platform (not 32-bit, not 64-bit?)
+#endif
+#endif
+
+#if ((defined(_WIN32) || defined(_WIN64)) && !defined(__clang__)) ||           \
+    (defined(_M_ARM64) && !defined(__MINGW32__))
+#include <intrin.h>
+#endif
+
+#if defined(_MSC_VER) && !defined(__clang__)
+#define SIMDJSON_FASTFLOAT_VISUAL_STUDIO 1
+#endif
+
+#if defined __BYTE_ORDER__ && defined __ORDER_BIG_ENDIAN__
+#define SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN (__BYTE_ORDER__ == __ORDER_BIG_ENDIAN__)
+#elif defined _WIN32
+#define SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN 0
+#else
+#if defined(__APPLE__) || defined(__FreeBSD__)
+#include <machine/endian.h>
+#elif defined(sun) || defined(__sun)
+#include <sys/byteorder.h>
+#elif defined(__MVS__)
+#include <sys/endian.h>
+#else
+#ifdef __has_include
+#if __has_include(<endian.h>)
+#include <endian.h>
+#endif //__has_include(<endian.h>)
+#endif //__has_include
+#endif
+#
+#ifndef __BYTE_ORDER__
+// safe choice
+#define SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN 0
+#endif
+#
+#ifndef __ORDER_LITTLE_ENDIAN__
+// safe choice
+#define SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN 0
+#endif
+#
+#if __BYTE_ORDER__ == __ORDER_LITTLE_ENDIAN__
+#define SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN 0
+#else
+#define SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN 1
+#endif
+#endif
+
+#if defined(__SSE2__) || (defined(SIMDJSON_FASTFLOAT_VISUAL_STUDIO) &&                  \
+                          (defined(_M_AMD64) || defined(_M_X64) ||             \
+                           (defined(_M_IX86_FP) && _M_IX86_FP == 2)))
+#define SIMDJSON_FASTFLOAT_SSE2 1
+#endif
+
+#if defined(__aarch64__) || defined(_M_ARM64)
+#define SIMDJSON_FASTFLOAT_NEON 1
+#endif
+
+#if defined(SIMDJSON_FASTFLOAT_SSE2) || defined(SIMDJSON_FASTFLOAT_NEON)
+#define SIMDJSON_FASTFLOAT_HAS_SIMD 1
+#endif
+
+#if defined(__GNUC__)
+// disable -Wcast-align=strict (GCC only)
+#define SIMDJSON_FASTFLOAT_SIMD_DISABLE_WARNINGS                                        \
+  _Pragma("GCC diagnostic push")                                               \
+      _Pragma("GCC diagnostic ignored \"-Wcast-align\"")
+#else
+#define SIMDJSON_FASTFLOAT_SIMD_DISABLE_WARNINGS
+#endif
+
+#if defined(__GNUC__)
+#define SIMDJSON_FASTFLOAT_SIMD_RESTORE_WARNINGS _Pragma("GCC diagnostic pop")
+#else
+#define SIMDJSON_FASTFLOAT_SIMD_RESTORE_WARNINGS
+#endif
+
+#ifdef SIMDJSON_FASTFLOAT_VISUAL_STUDIO
+#define simdjson_fastfloat_really_inline __forceinline
+#else
+#define simdjson_fastfloat_really_inline inline __attribute__((always_inline))
+#endif
+
+// Branch-probability hint marking the rare slow-path branches as cold, so the
+// optimizer keeps the out-of-line slow-path re-parse off the hot path (and does
+// not duplicate the force-inlined hot scanner into the caller, which bloated
+// the hot frame and hurt ILP on some targets). Used at the call site as
+//   if simdjson_fastfloat_unlikely(cond) { ... }
+// (the macro supplies the parentheses). It expands to the standard [[unlikely]]
+// attribute when supported, otherwise to __builtin_expect on GCC/Clang, or
+// to a no-op elsewhere (e.g. pre-C++20 MSVC, which has no equivalent hint).
+#ifdef __has_cpp_attribute
+#if __has_cpp_attribute(unlikely) >= 201803L
+// g++-9 hits hits this branch, but then fails to compile
+// [[unlikely]]. This happens only with g++-9.
+#if !defined(__GNUC__) || (__GNUC__ != 9)
+#define SIMDJSON_FASTFLOAT_USE_UNLIKELY_ATTR
+#endif
+#endif
+#endif
+
+#ifdef SIMDJSON_FASTFLOAT_USE_UNLIKELY_ATTR
+#define simdjson_fastfloat_unlikely(x) (x) [[unlikely]]
+#elif defined(__GNUC__) || defined(__clang__)
+#define simdjson_fastfloat_unlikely(x) (__builtin_expect(!!(x), 0))
+#else
+#define simdjson_fastfloat_unlikely(x) (x)
+#endif
+
+#ifndef SIMDJSON_FASTFLOAT_ASSERT
+#define SIMDJSON_FASTFLOAT_ASSERT(x)                                                    \
+  {                                                                            \
+    static_cast<void>(x);                                                      \
+  }
+#endif
+
+#ifndef SIMDJSON_FASTFLOAT_DEBUG_ASSERT
+#define SIMDJSON_FASTFLOAT_DEBUG_ASSERT(x)                                              \
+  {                                                                            \
+    static_cast<void>(x);                                                      \
+  }
+#endif
+
+// rust style `try!()` macro, or `?` operator
+#define SIMDJSON_FASTFLOAT_TRY(x)                                                       \
+  {                                                                            \
+    if (!(x))                                                                  \
+      return false;                                                            \
+  }
+
+#define SIMDJSON_FASTFLOAT_ENABLE_IF(...)                                               \
+  typename std::enable_if<(__VA_ARGS__), int>::type
+
+namespace simdjson_fast_float {
+
+simdjson_fastfloat_really_inline constexpr bool cpp20_and_in_constexpr() {
+#if SIMDJSON_FASTFLOAT_HAS_IS_CONSTANT_EVALUATED
+  return std::is_constant_evaluated();
+#else
+  return false;
+#endif
+}
+
+template <typename T>
+struct is_supported_float_type
+    : std::integral_constant<
+          bool, std::is_same<T, double>::value || std::is_same<T, float>::value
+#ifdef __STDCPP_FLOAT64_T__
+                    || std::is_same<T, std::float64_t>::value
+#endif
+#ifdef __STDCPP_FLOAT32_T__
+                    || std::is_same<T, std::float32_t>::value
+#endif
+#ifdef __STDCPP_FLOAT16_T__
+                    || std::is_same<T, std::float16_t>::value
+#endif
+#ifdef __STDCPP_BFLOAT16_T__
+                    || std::is_same<T, std::bfloat16_t>::value
+#endif
+          > {
+};
+
+template <typename T>
+using equiv_uint_t = typename std::conditional<
+    sizeof(T) == 1, uint8_t,
+    typename std::conditional<
+        sizeof(T) == 2, uint16_t,
+        typename std::conditional<sizeof(T) == 4, uint32_t,
+                                  uint64_t>::type>::type>::type;
+
+template <typename T> struct is_supported_integer_type : std::is_integral<T> {};
+
+template <typename UC>
+struct is_supported_char_type
+    : std::integral_constant<bool, std::is_same<UC, char>::value ||
+                                       std::is_same<UC, wchar_t>::value ||
+                                       std::is_same<UC, char16_t>::value ||
+                                       std::is_same<UC, char32_t>::value
+#ifdef __cpp_char8_t
+                                       || std::is_same<UC, char8_t>::value
+#endif
+                             > {
+};
+
+template <typename UC>
+inline SIMDJSON_FASTFLOAT_CONSTEXPR14 bool
+simdjson_fastfloat_strncasecmp3(UC const *actual_mixedcase,
+                       UC const *expected_lowercase) {
+  uint64_t mask{0};
+  SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 1) { mask = 0x2020202020202020; }
+  else SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 2) {
+    mask = 0x0020002000200020;
+  }
+  else SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 4) {
+    mask = 0x0000002000000020;
+  }
+  else {
+    return false;
+  }
+
+  uint64_t val1{0}, val2{0};
+  if (cpp20_and_in_constexpr()) {
+    for (size_t i = 0; i < 3; i++) {
+      if ((actual_mixedcase[i] | 32) != expected_lowercase[i]) {
+        return false;
+      }
+    }
+    return true;
+  } else {
+    SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 1 || sizeof(UC) == 2) {
+      ::memcpy(&val1, actual_mixedcase, 3 * sizeof(UC));
+      ::memcpy(&val2, expected_lowercase, 3 * sizeof(UC));
+      val1 |= mask;
+      val2 |= mask;
+      return val1 == val2;
+    }
+    else SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 4) {
+      ::memcpy(&val1, actual_mixedcase, 2 * sizeof(UC));
+      ::memcpy(&val2, expected_lowercase, 2 * sizeof(UC));
+      val1 |= mask;
+      if (val1 != val2) {
+        return false;
+      }
+      return (actual_mixedcase[2] | 32) == (expected_lowercase[2]);
+    }
+    else {
+      return false;
+    }
+  }
+}
+
+template <typename UC>
+inline SIMDJSON_FASTFLOAT_CONSTEXPR14 bool
+simdjson_fastfloat_strncasecmp5(UC const *actual_mixedcase,
+                       UC const *expected_lowercase) {
+  uint64_t mask{0};
+  uint64_t val1{0}, val2{0};
+  if (cpp20_and_in_constexpr()) {
+    for (size_t i = 0; i < 5; i++) {
+      if ((actual_mixedcase[i] | 32) != expected_lowercase[i]) {
+        return false;
+      }
+    }
+    return true;
+  } else {
+    SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 1) {
+      mask = 0x2020202020202020;
+      ::memcpy(&val1, actual_mixedcase, 5 * sizeof(UC));
+      ::memcpy(&val2, expected_lowercase, 5 * sizeof(UC));
+      val1 |= mask;
+      val2 |= mask;
+      return val1 == val2;
+    }
+    else SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 2) {
+      mask = 0x0020002000200020;
+      ::memcpy(&val1, actual_mixedcase, 4 * sizeof(UC));
+      ::memcpy(&val2, expected_lowercase, 4 * sizeof(UC));
+      val1 |= mask;
+      if (val1 != val2) {
+        return false;
+      }
+      return (actual_mixedcase[4] | 32) == (expected_lowercase[4]);
+    }
+    else SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 4) {
+      mask = 0x0000002000000020;
+      ::memcpy(&val1, actual_mixedcase, 2 * sizeof(UC));
+      ::memcpy(&val2, expected_lowercase, 2 * sizeof(UC));
+      val1 |= mask;
+      if (val1 != val2) {
+        return false;
+      }
+      ::memcpy(&val1, actual_mixedcase + 2, 2 * sizeof(UC));
+      ::memcpy(&val2, expected_lowercase + 2, 2 * sizeof(UC));
+      val1 |= mask;
+      if (val1 != val2) {
+        return false;
+      }
+      return (actual_mixedcase[4] | 32) == (expected_lowercase[4]);
+    }
+    else {
+      return false;
+    }
+  }
+}
+
+// Compares two ASCII strings in a case insensitive manner.
+template <typename UC>
+inline SIMDJSON_FASTFLOAT_CONSTEXPR14 bool
+simdjson_fastfloat_strncasecmp(UC const *actual_mixedcase, UC const *expected_lowercase,
+                      size_t length) {
+  uint64_t mask{0};
+  SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 1) { mask = 0x2020202020202020; }
+  else SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 2) {
+    mask = 0x0020002000200020;
+  }
+  else SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 4) {
+    mask = 0x0000002000000020;
+  }
+  else {
+    return false;
+  }
+
+  if (cpp20_and_in_constexpr()) {
+    for (size_t i = 0; i < length; i++) {
+      if ((actual_mixedcase[i] | 32) != expected_lowercase[i]) {
+        return false;
+      }
+    }
+    return true;
+  } else {
+    uint64_t val1{0}, val2{0};
+    size_t sz{8 / (sizeof(UC))};
+    for (size_t i = 0; i < length; i += sz) {
+      val1 = val2 = 0;
+      sz = sz < (length - i) ? sz : length - i;
+      ::memcpy(&val1, actual_mixedcase + i, sz * sizeof(UC));
+      ::memcpy(&val2, expected_lowercase + i, sz * sizeof(UC));
+      val1 |= mask;
+      val2 |= mask;
+      if (val1 != val2) {
+        return false;
+      }
+    }
+    return true;
+  }
+}
+
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+
+// a pointer and a length to a contiguous block of memory
+template <typename T> struct span {
+  T const *ptr;
+  size_t length;
+
+  constexpr span(T const *_ptr, size_t _length) : ptr(_ptr), length(_length) {}
+
+  constexpr span() : ptr(nullptr), length(0) {}
+
+  constexpr size_t len() const noexcept { return length; }
+
+  SIMDJSON_FASTFLOAT_CONSTEXPR14 const T &operator[](size_t index) const noexcept {
+    SIMDJSON_FASTFLOAT_DEBUG_ASSERT(index < length);
+    return ptr[index];
+  }
+};
+
+struct value128 {
+  uint64_t low;
+  uint64_t high;
+
+  constexpr value128(uint64_t _low, uint64_t _high) : low(_low), high(_high) {}
+
+  constexpr value128() : low(0), high(0) {}
+};
+
+/* Helper C++14 constexpr generic implementation of leading_zeroes */
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 int
+leading_zeroes_generic(uint64_t input_num, int last_bit = 0) {
+  if (input_num & uint64_t(0xffffffff00000000)) {
+    input_num >>= 32;
+    last_bit |= 32;
+  }
+  if (input_num & uint64_t(0xffff0000)) {
+    input_num >>= 16;
+    last_bit |= 16;
+  }
+  if (input_num & uint64_t(0xff00)) {
+    input_num >>= 8;
+    last_bit |= 8;
+  }
+  if (input_num & uint64_t(0xf0)) {
+    input_num >>= 4;
+    last_bit |= 4;
+  }
+  if (input_num & uint64_t(0xc)) {
+    input_num >>= 2;
+    last_bit |= 2;
+  }
+  if (input_num & uint64_t(0x2)) { /* input_num >>=  1; */
+    last_bit |= 1;
+  }
+  return 63 - last_bit;
+}
+
+/* result might be undefined when input_num is zero */
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 int
+leading_zeroes(uint64_t input_num) {
+  assert(input_num > 0);
+  if (cpp20_and_in_constexpr()) {
+    return leading_zeroes_generic(input_num);
+  }
+#ifdef SIMDJSON_FASTFLOAT_VISUAL_STUDIO
+#if defined(_M_X64) || defined(_M_ARM64)
+  unsigned long leading_zero = 0;
+  // Search the mask data from most significant bit (MSB)
+  // to least significant bit (LSB) for a set bit (1).
+  _BitScanReverse64(&leading_zero, input_num);
+  return static_cast<int>(63 - leading_zero);
+#else
+  return leading_zeroes_generic(input_num);
+#endif
+#else
+  return __builtin_clzll(input_num);
+#endif
+}
+
+/* Helper C++14 constexpr generic implementation of countr_zero for 32-bit */
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 int
+countr_zero_generic_32(uint32_t input_num) {
+  if (input_num == 0) {
+    return 32;
+  }
+  int last_bit = 0;
+  if (!(input_num & 0x0000FFFF)) {
+    input_num >>= 16;
+    last_bit |= 16;
+  }
+  if (!(input_num & 0x00FF)) {
+    input_num >>= 8;
+    last_bit |= 8;
+  }
+  if (!(input_num & 0x0F)) {
+    input_num >>= 4;
+    last_bit |= 4;
+  }
+  if (!(input_num & 0x3)) {
+    input_num >>= 2;
+    last_bit |= 2;
+  }
+  if (!(input_num & 0x1)) {
+    last_bit |= 1;
+  }
+  return last_bit;
+}
+
+/* count trailing zeroes for 32-bit integers */
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 int
+countr_zero_32(uint32_t input_num) {
+  if (cpp20_and_in_constexpr()) {
+    return countr_zero_generic_32(input_num);
+  }
+#ifdef SIMDJSON_FASTFLOAT_VISUAL_STUDIO
+  unsigned long trailing_zero = 0;
+  if (_BitScanForward(&trailing_zero, input_num)) {
+    return static_cast<int>(trailing_zero);
+  }
+  return 32;
+#else
+  return input_num == 0 ? 32 : __builtin_ctz(input_num);
+#endif
+}
+
+// slow emulation routine for 32-bit
+simdjson_fastfloat_really_inline constexpr uint64_t emulu(uint32_t x, uint32_t y) {
+  return x * static_cast<uint64_t>(y);
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 uint64_t
+umul128_generic(uint64_t ab, uint64_t cd, uint64_t *hi) {
+  uint64_t ad =
+      emulu(static_cast<uint32_t>(ab >> 32), static_cast<uint32_t>(cd));
+  uint64_t bd = emulu(static_cast<uint32_t>(ab), static_cast<uint32_t>(cd));
+  uint64_t adbc =
+      ad + emulu(static_cast<uint32_t>(ab), static_cast<uint32_t>(cd >> 32));
+  uint64_t adbc_carry = static_cast<uint64_t>(adbc < ad);
+  uint64_t lo = bd + (adbc << 32);
+  *hi =
+      emulu(static_cast<uint32_t>(ab >> 32), static_cast<uint32_t>(cd >> 32)) +
+      (adbc >> 32) + (adbc_carry << 32) + static_cast<uint64_t>(lo < bd);
+  return lo;
+}
+
+#ifdef SIMDJSON_FASTFLOAT_32BIT
+
+// slow emulation routine for 32-bit
+#if !defined(__MINGW64__)
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 uint64_t _umul128(uint64_t ab,
+                                                                uint64_t cd,
+                                                                uint64_t *hi) {
+  return umul128_generic(ab, cd, hi);
+}
+#endif // !__MINGW64__
+
+#endif // SIMDJSON_FASTFLOAT_32BIT
+
+// compute 64-bit a*b
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 value128
+full_multiplication(uint64_t a, uint64_t b) {
+  if (cpp20_and_in_constexpr()) {
+    value128 answer;
+    answer.low = umul128_generic(a, b, &answer.high);
+    return answer;
+  }
+  value128 answer;
+#if defined(_M_ARM64) && !defined(__MINGW32__)
+  // ARM64 has native support for 64-bit multiplications, no need to emulate
+  // But MinGW on ARM64 doesn't have native support for 64-bit multiplications
+  answer.high = __umulh(a, b);
+  answer.low = a * b;
+#elif defined(SIMDJSON_FASTFLOAT_32BIT) || (defined(_WIN64) && !defined(__clang__) &&   \
+                                   !defined(_M_ARM64) && !defined(__GNUC__))
+  answer.low = _umul128(a, b, &answer.high); // _umul128 not available on ARM64
+#elif defined(SIMDJSON_FASTFLOAT_64BIT) && defined(__SIZEOF_INT128__)
+  __uint128_t r = static_cast<__uint128_t>(a) * b;
+  answer.low = uint64_t(r);
+  answer.high = uint64_t(r >> 64);
+#else
+  answer.low = umul128_generic(a, b, &answer.high);
+#endif
+  return answer;
+}
+
+struct adjusted_mantissa {
+  uint64_t mantissa{0};
+  int32_t power2{0}; // a negative value indicates an invalid result
+  adjusted_mantissa() = default;
+
+  constexpr bool operator==(adjusted_mantissa const &o) const {
+    return mantissa == o.mantissa && power2 == o.power2;
+  }
+
+  constexpr bool operator!=(adjusted_mantissa const &o) const {
+    return mantissa != o.mantissa || power2 != o.power2;
+  }
+};
+
+// Bias so we can get the real exponent with an invalid adjusted_mantissa.
+constexpr static int32_t invalid_am_bias = -0x8000;
+
+// used for binary_format_lookup_tables<T>::max_mantissa
+constexpr uint64_t constant_55555 = 5 * 5 * 5 * 5 * 5;
+
+template <typename T, typename U = void> struct binary_format_lookup_tables;
+
+template <typename T> struct binary_format : binary_format_lookup_tables<T> {
+  using equiv_uint = equiv_uint_t<T>;
+
+  static constexpr int mantissa_explicit_bits();
+  static constexpr int minimum_exponent();
+  static constexpr int infinite_power();
+  static constexpr int sign_index();
+  static constexpr int
+  min_exponent_fast_path(); // used when fegetround() == FE_TONEAREST
+  static constexpr int max_exponent_fast_path();
+  static constexpr int max_exponent_round_to_even();
+  static constexpr int min_exponent_round_to_even();
+  static constexpr uint64_t max_mantissa_fast_path(int64_t power);
+  static constexpr uint64_t
+  max_mantissa_fast_path(); // used when fegetround() == FE_TONEAREST
+  static constexpr int largest_power_of_ten();
+  static constexpr int smallest_power_of_ten();
+  static constexpr T exact_power_of_ten(int64_t power);
+  static constexpr size_t max_digits();
+  static constexpr equiv_uint exponent_mask();
+  static constexpr equiv_uint mantissa_mask();
+  static constexpr equiv_uint hidden_bit_mask();
+};
+
+template <typename U> struct binary_format_lookup_tables<double, U> {
+  static constexpr double powers_of_ten[] = {
+      1e0,  1e1,  1e2,  1e3,  1e4,  1e5,  1e6,  1e7,  1e8,  1e9,  1e10, 1e11,
+      1e12, 1e13, 1e14, 1e15, 1e16, 1e17, 1e18, 1e19, 1e20, 1e21, 1e22};
+
+  // Largest integer value v so that (5**index * v) <= 1<<53.
+  // 0x20000000000000 == 1 << 53
+  static constexpr uint64_t max_mantissa[] = {
+      0x20000000000000,
+      0x20000000000000 / 5,
+      0x20000000000000 / (5 * 5),
+      0x20000000000000 / (5 * 5 * 5),
+      0x20000000000000 / (5 * 5 * 5 * 5),
+      0x20000000000000 / (constant_55555),
+      0x20000000000000 / (constant_55555 * 5),
+      0x20000000000000 / (constant_55555 * 5 * 5),
+      0x20000000000000 / (constant_55555 * 5 * 5 * 5),
+      0x20000000000000 / (constant_55555 * 5 * 5 * 5 * 5),
+      0x20000000000000 / (constant_55555 * constant_55555),
+      0x20000000000000 / (constant_55555 * constant_55555 * 5),
+      0x20000000000000 / (constant_55555 * constant_55555 * 5 * 5),
+      0x20000000000000 / (constant_55555 * constant_55555 * 5 * 5 * 5),
+      0x20000000000000 / (constant_55555 * constant_55555 * constant_55555),
+      0x20000000000000 / (constant_55555 * constant_55555 * constant_55555 * 5),
+      0x20000000000000 /
+          (constant_55555 * constant_55555 * constant_55555 * 5 * 5),
+      0x20000000000000 /
+          (constant_55555 * constant_55555 * constant_55555 * 5 * 5 * 5),
+      0x20000000000000 /
+          (constant_55555 * constant_55555 * constant_55555 * 5 * 5 * 5 * 5),
+      0x20000000000000 /
+          (constant_55555 * constant_55555 * constant_55555 * constant_55555),
+      0x20000000000000 / (constant_55555 * constant_55555 * constant_55555 *
+                          constant_55555 * 5),
+      0x20000000000000 / (constant_55555 * constant_55555 * constant_55555 *
+                          constant_55555 * 5 * 5),
+      0x20000000000000 / (constant_55555 * constant_55555 * constant_55555 *
+                          constant_55555 * 5 * 5 * 5),
+      0x20000000000000 / (constant_55555 * constant_55555 * constant_55555 *
+                          constant_55555 * 5 * 5 * 5 * 5)};
+};
+
+#if SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE
+
+template <typename U>
+constexpr double binary_format_lookup_tables<double, U>::powers_of_ten[];
+
+template <typename U>
+constexpr uint64_t binary_format_lookup_tables<double, U>::max_mantissa[];
+
+#endif
+
+template <typename U> struct binary_format_lookup_tables<float, U> {
+  static constexpr float powers_of_ten[] = {1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+                                            1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+
+  // Largest integer value v so that (5**index * v) <= 1<<24.
+  // 0x1000000 == 1<<24
+  static constexpr uint64_t max_mantissa[] = {
+      0x1000000,
+      0x1000000 / 5,
+      0x1000000 / (5 * 5),
+      0x1000000 / (5 * 5 * 5),
+      0x1000000 / (5 * 5 * 5 * 5),
+      0x1000000 / (constant_55555),
+      0x1000000 / (constant_55555 * 5),
+      0x1000000 / (constant_55555 * 5 * 5),
+      0x1000000 / (constant_55555 * 5 * 5 * 5),
+      0x1000000 / (constant_55555 * 5 * 5 * 5 * 5),
+      0x1000000 / (constant_55555 * constant_55555),
+      0x1000000 / (constant_55555 * constant_55555 * 5)};
+};
+
+#if SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE
+
+template <typename U>
+constexpr float binary_format_lookup_tables<float, U>::powers_of_ten[];
+
+template <typename U>
+constexpr uint64_t binary_format_lookup_tables<float, U>::max_mantissa[];
+
+#endif
+
+template <>
+inline constexpr int binary_format<double>::min_exponent_fast_path() {
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+  return 0;
+#else
+  return -22;
+#endif
+}
+
+template <>
+inline constexpr int binary_format<float>::min_exponent_fast_path() {
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+  return 0;
+#else
+  return -10;
+#endif
+}
+
+template <>
+inline constexpr int binary_format<double>::mantissa_explicit_bits() {
+  return 52;
+}
+
+template <>
+inline constexpr int binary_format<float>::mantissa_explicit_bits() {
+  return 23;
+}
+
+template <>
+inline constexpr int binary_format<double>::max_exponent_round_to_even() {
+  return 23;
+}
+
+template <>
+inline constexpr int binary_format<float>::max_exponent_round_to_even() {
+  return 10;
+}
+
+template <>
+inline constexpr int binary_format<double>::min_exponent_round_to_even() {
+  return -4;
+}
+
+template <>
+inline constexpr int binary_format<float>::min_exponent_round_to_even() {
+  return -17;
+}
+
+template <> inline constexpr int binary_format<double>::minimum_exponent() {
+  return -1023;
+}
+
+template <> inline constexpr int binary_format<float>::minimum_exponent() {
+  return -127;
+}
+
+template <> inline constexpr int binary_format<double>::infinite_power() {
+  return 0x7FF;
+}
+
+template <> inline constexpr int binary_format<float>::infinite_power() {
+  return 0xFF;
+}
+
+template <> inline constexpr int binary_format<double>::sign_index() {
+  return 63;
+}
+
+template <> inline constexpr int binary_format<float>::sign_index() {
+  return 31;
+}
+
+template <>
+inline constexpr int binary_format<double>::max_exponent_fast_path() {
+  return 22;
+}
+
+template <>
+inline constexpr int binary_format<float>::max_exponent_fast_path() {
+  return 10;
+}
+
+template <>
+inline constexpr uint64_t binary_format<double>::max_mantissa_fast_path() {
+  return uint64_t(2) << mantissa_explicit_bits();
+}
+
+template <>
+inline constexpr uint64_t binary_format<float>::max_mantissa_fast_path() {
+  return uint64_t(2) << mantissa_explicit_bits();
+}
+
+// credit: Jakub Jelinek
+#ifdef __STDCPP_FLOAT16_T__
+template <typename U> struct binary_format_lookup_tables<std::float16_t, U> {
+  static constexpr std::float16_t powers_of_ten[] = {1e0f16, 1e1f16, 1e2f16,
+                                                     1e3f16, 1e4f16};
+
+  // Largest integer value v so that (5**index * v) <= 1<<11.
+  // 0x800 == 1<<11
+  static constexpr uint64_t max_mantissa[] = {0x800,
+                                              0x800 / 5,
+                                              0x800 / (5 * 5),
+                                              0x800 / (5 * 5 * 5),
+                                              0x800 / (5 * 5 * 5 * 5),
+                                              0x800 / (constant_55555)};
+};
+
+#if SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE
+
+template <typename U>
+constexpr std::float16_t
+    binary_format_lookup_tables<std::float16_t, U>::powers_of_ten[];
+
+template <typename U>
+constexpr uint64_t
+    binary_format_lookup_tables<std::float16_t, U>::max_mantissa[];
+
+#endif
+
+template <>
+inline constexpr std::float16_t
+binary_format<std::float16_t>::exact_power_of_ten(int64_t power) {
+  // Work around clang bug https://godbolt.org/z/zedh7rrhc
+  return static_cast<void>(powers_of_ten[0]), powers_of_ten[power];
+}
+
+template <>
+inline constexpr binary_format<std::float16_t>::equiv_uint
+binary_format<std::float16_t>::exponent_mask() {
+  return 0x7C00;
+}
+
+template <>
+inline constexpr binary_format<std::float16_t>::equiv_uint
+binary_format<std::float16_t>::mantissa_mask() {
+  return 0x03FF;
+}
+
+template <>
+inline constexpr binary_format<std::float16_t>::equiv_uint
+binary_format<std::float16_t>::hidden_bit_mask() {
+  return 0x0400;
+}
+
+template <>
+inline constexpr int binary_format<std::float16_t>::max_exponent_fast_path() {
+  return 4;
+}
+
+template <>
+inline constexpr int binary_format<std::float16_t>::mantissa_explicit_bits() {
+  return 10;
+}
+
+template <>
+inline constexpr uint64_t
+binary_format<std::float16_t>::max_mantissa_fast_path() {
+  return uint64_t(2) << mantissa_explicit_bits();
+}
+
+template <>
+inline constexpr uint64_t
+binary_format<std::float16_t>::max_mantissa_fast_path(int64_t power) {
+  // caller is responsible to ensure that
+  // power >= 0 && power <= 4
+  //
+  // Work around clang bug https://godbolt.org/z/zedh7rrhc
+  return static_cast<void>(max_mantissa[0]), max_mantissa[power];
+}
+
+template <>
+inline constexpr int binary_format<std::float16_t>::min_exponent_fast_path() {
+  return 0;
+}
+
+template <>
+inline constexpr int
+binary_format<std::float16_t>::max_exponent_round_to_even() {
+  return 5;
+}
+
+template <>
+inline constexpr int
+binary_format<std::float16_t>::min_exponent_round_to_even() {
+  return -22;
+}
+
+template <>
+inline constexpr int binary_format<std::float16_t>::minimum_exponent() {
+  return -15;
+}
+
+template <>
+inline constexpr int binary_format<std::float16_t>::infinite_power() {
+  return 0x1F;
+}
+
+template <> inline constexpr int binary_format<std::float16_t>::sign_index() {
+  return 15;
+}
+
+template <>
+inline constexpr int binary_format<std::float16_t>::largest_power_of_ten() {
+  return 4;
+}
+
+template <>
+inline constexpr int binary_format<std::float16_t>::smallest_power_of_ten() {
+  return -27;
+}
+
+template <>
+inline constexpr size_t binary_format<std::float16_t>::max_digits() {
+  return 22;
+}
+#endif // __STDCPP_FLOAT16_T__
+
+// credit: Jakub Jelinek
+#ifdef __STDCPP_BFLOAT16_T__
+template <typename U> struct binary_format_lookup_tables<std::bfloat16_t, U> {
+  static constexpr std::bfloat16_t powers_of_ten[] = {1e0bf16, 1e1bf16, 1e2bf16,
+                                                      1e3bf16};
+
+  // Largest integer value v so that (5**index * v) <= 1<<8.
+  // 0x100 == 1<<8
+  static constexpr uint64_t max_mantissa[] = {0x100, 0x100 / 5, 0x100 / (5 * 5),
+                                              0x100 / (5 * 5 * 5),
+                                              0x100 / (5 * 5 * 5 * 5)};
+};
+
+#if SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE
+
+template <typename U>
+constexpr std::bfloat16_t
+    binary_format_lookup_tables<std::bfloat16_t, U>::powers_of_ten[];
+
+template <typename U>
+constexpr uint64_t
+    binary_format_lookup_tables<std::bfloat16_t, U>::max_mantissa[];
+
+#endif
+
+template <>
+inline constexpr std::bfloat16_t
+binary_format<std::bfloat16_t>::exact_power_of_ten(int64_t power) {
+  // Work around clang bug https://godbolt.org/z/zedh7rrhc
+  return static_cast<void>(powers_of_ten[0]), powers_of_ten[power];
+}
+
+template <>
+inline constexpr int binary_format<std::bfloat16_t>::max_exponent_fast_path() {
+  return 3;
+}
+
+template <>
+inline constexpr binary_format<std::bfloat16_t>::equiv_uint
+binary_format<std::bfloat16_t>::exponent_mask() {
+  return 0x7F80;
+}
+
+template <>
+inline constexpr binary_format<std::bfloat16_t>::equiv_uint
+binary_format<std::bfloat16_t>::mantissa_mask() {
+  return 0x007F;
+}
+
+template <>
+inline constexpr binary_format<std::bfloat16_t>::equiv_uint
+binary_format<std::bfloat16_t>::hidden_bit_mask() {
+  return 0x0080;
+}
+
+template <>
+inline constexpr int binary_format<std::bfloat16_t>::mantissa_explicit_bits() {
+  return 7;
+}
+
+template <>
+inline constexpr uint64_t
+binary_format<std::bfloat16_t>::max_mantissa_fast_path() {
+  return uint64_t(2) << mantissa_explicit_bits();
+}
+
+template <>
+inline constexpr uint64_t
+binary_format<std::bfloat16_t>::max_mantissa_fast_path(int64_t power) {
+  // caller is responsible to ensure that
+  // power >= 0 && power <= 3
+  //
+  // Work around clang bug https://godbolt.org/z/zedh7rrhc
+  return static_cast<void>(max_mantissa[0]), max_mantissa[power];
+}
+
+template <>
+inline constexpr int binary_format<std::bfloat16_t>::min_exponent_fast_path() {
+  return 0;
+}
+
+template <>
+inline constexpr int
+binary_format<std::bfloat16_t>::max_exponent_round_to_even() {
+  return 3;
+}
+
+template <>
+inline constexpr int
+binary_format<std::bfloat16_t>::min_exponent_round_to_even() {
+  return -24;
+}
+
+template <>
+inline constexpr int binary_format<std::bfloat16_t>::minimum_exponent() {
+  return -127;
+}
+
+template <>
+inline constexpr int binary_format<std::bfloat16_t>::infinite_power() {
+  return 0xFF;
+}
+
+template <> inline constexpr int binary_format<std::bfloat16_t>::sign_index() {
+  return 15;
+}
+
+template <>
+inline constexpr int binary_format<std::bfloat16_t>::largest_power_of_ten() {
+  return 38;
+}
+
+template <>
+inline constexpr int binary_format<std::bfloat16_t>::smallest_power_of_ten() {
+  return -60;
+}
+
+template <>
+inline constexpr size_t binary_format<std::bfloat16_t>::max_digits() {
+  return 98;
+}
+#endif // __STDCPP_BFLOAT16_T__
+
+template <>
+inline constexpr uint64_t
+binary_format<double>::max_mantissa_fast_path(int64_t power) {
+  // caller is responsible to ensure that
+  // power >= 0 && power <= 22
+  //
+  // Work around clang bug https://godbolt.org/z/zedh7rrhc
+  return static_cast<void>(max_mantissa[0]), max_mantissa[power];
+}
+
+template <>
+inline constexpr uint64_t
+binary_format<float>::max_mantissa_fast_path(int64_t power) {
+  // caller is responsible to ensure that
+  // power >= 0 && power <= 10
+  //
+  // Work around clang bug https://godbolt.org/z/zedh7rrhc
+  return static_cast<void>(max_mantissa[0]), max_mantissa[power];
+}
+
+template <>
+inline constexpr double
+binary_format<double>::exact_power_of_ten(int64_t power) {
+  // Work around clang bug https://godbolt.org/z/zedh7rrhc
+  return static_cast<void>(powers_of_ten[0]), powers_of_ten[power];
+}
+
+template <>
+inline constexpr float binary_format<float>::exact_power_of_ten(int64_t power) {
+  // Work around clang bug https://godbolt.org/z/zedh7rrhc
+  return static_cast<void>(powers_of_ten[0]), powers_of_ten[power];
+}
+
+template <> inline constexpr int binary_format<double>::largest_power_of_ten() {
+  return 308;
+}
+
+template <> inline constexpr int binary_format<float>::largest_power_of_ten() {
+  return 38;
+}
+
+template <>
+inline constexpr int binary_format<double>::smallest_power_of_ten() {
+  return -342;
+}
+
+template <> inline constexpr int binary_format<float>::smallest_power_of_ten() {
+  return -64;
+}
+
+template <> inline constexpr size_t binary_format<double>::max_digits() {
+  return 769;
+}
+
+template <> inline constexpr size_t binary_format<float>::max_digits() {
+  return 114;
+}
+
+template <>
+inline constexpr binary_format<float>::equiv_uint
+binary_format<float>::exponent_mask() {
+  return 0x7F800000;
+}
+
+template <>
+inline constexpr binary_format<double>::equiv_uint
+binary_format<double>::exponent_mask() {
+  return 0x7FF0000000000000;
+}
+
+template <>
+inline constexpr binary_format<float>::equiv_uint
+binary_format<float>::mantissa_mask() {
+  return 0x007FFFFF;
+}
+
+template <>
+inline constexpr binary_format<double>::equiv_uint
+binary_format<double>::mantissa_mask() {
+  return 0x000FFFFFFFFFFFFF;
+}
+
+template <>
+inline constexpr binary_format<float>::equiv_uint
+binary_format<float>::hidden_bit_mask() {
+  return 0x00800000;
+}
+
+template <>
+inline constexpr binary_format<double>::equiv_uint
+binary_format<double>::hidden_bit_mask() {
+  return 0x0010000000000000;
+}
+
+template <typename T>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 void
+to_float(bool negative, adjusted_mantissa am, T &value) {
+  using equiv_uint = equiv_uint_t<T>;
+  equiv_uint word = equiv_uint(am.mantissa);
+  word = equiv_uint(word | equiv_uint(am.power2)
+                               << binary_format<T>::mantissa_explicit_bits());
+  word =
+      equiv_uint(word | equiv_uint(negative) << binary_format<T>::sign_index());
+#if SIMDJSON_FASTFLOAT_HAS_BIT_CAST
+  value = std::bit_cast<T>(word);
+#else
+  ::memcpy(&value, &word, sizeof(T));
+#endif
+}
+
+template <typename = void> struct space_lut {
+  static constexpr bool value[] = {
+      0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+      0, 0, 0, 0, 0, 0, 0, 0, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+      0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+      0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+      0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+      0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+      0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+      0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+      0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+      0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+      0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0};
+};
+
+#if SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE
+
+template <typename T> constexpr bool space_lut<T>::value[];
+
+#endif
+
+template <typename UC> constexpr bool is_space(UC c) {
+  // wchar_t and char can be signed, so a negative code unit slips past a plain
+  // `c < 256` and then indexes the table by its truncated low byte. Compare as
+  // unsigned, matching the care taken in ch_to_digit.
+  using UnsignedUC = typename std::make_unsigned<UC>::type;
+  return static_cast<UnsignedUC>(c) < 256 && space_lut<>::value[uint8_t(c)];
+}
+
+template <typename UC> static constexpr uint64_t int_cmp_zeros() {
+  static_assert((sizeof(UC) == 1) || (sizeof(UC) == 2) || (sizeof(UC) == 4),
+                "Unsupported character size");
+  return (sizeof(UC) == 1) ? 0x3030303030303030
+         : (sizeof(UC) == 2)
+             ? (uint64_t(UC('0')) << 48 | uint64_t(UC('0')) << 32 |
+                uint64_t(UC('0')) << 16 | UC('0'))
+             : (uint64_t(UC('0')) << 32 | UC('0'));
+}
+
+template <typename UC> static constexpr int int_cmp_len() {
+  return sizeof(uint64_t) / sizeof(UC);
+}
+
+template <typename UC> constexpr UC const *str_const_nan();
+
+template <> constexpr char const *str_const_nan<char>() { return "nan"; }
+
+template <> constexpr wchar_t const *str_const_nan<wchar_t>() { return L"nan"; }
+
+template <> constexpr char16_t const *str_const_nan<char16_t>() {
+  return u"nan";
+}
+
+template <> constexpr char32_t const *str_const_nan<char32_t>() {
+  return U"nan";
+}
+
+#ifdef __cpp_char8_t
+template <> constexpr char8_t const *str_const_nan<char8_t>() {
+  return u8"nan";
+}
+#endif
+
+template <typename UC> constexpr UC const *str_const_inf();
+
+template <> constexpr char const *str_const_inf<char>() { return "infinity"; }
+
+template <> constexpr wchar_t const *str_const_inf<wchar_t>() {
+  return L"infinity";
+}
+
+template <> constexpr char16_t const *str_const_inf<char16_t>() {
+  return u"infinity";
+}
+
+template <> constexpr char32_t const *str_const_inf<char32_t>() {
+  return U"infinity";
+}
+
+#ifdef __cpp_char8_t
+template <> constexpr char8_t const *str_const_inf<char8_t>() {
+  return u8"infinity";
+}
+#endif
+
+template <typename = void> struct int_luts {
+  static constexpr uint8_t chdigit[] = {
+      255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+      255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+      255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+      255, 255, 255, 0,   1,   2,   3,   4,   5,   6,   7,   8,   9,   255, 255,
+      255, 255, 255, 255, 255, 10,  11,  12,  13,  14,  15,  16,  17,  18,  19,
+      20,  21,  22,  23,  24,  25,  26,  27,  28,  29,  30,  31,  32,  33,  34,
+      35,  255, 255, 255, 255, 255, 255, 10,  11,  12,  13,  14,  15,  16,  17,
+      18,  19,  20,  21,  22,  23,  24,  25,  26,  27,  28,  29,  30,  31,  32,
+      33,  34,  35,  255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+      255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+      255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+      255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+      255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+      255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+      255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+      255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+      255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+      255};
+
+  static constexpr size_t maxdigits_u64[] = {
+      64, 41, 32, 28, 25, 23, 22, 21, 20, 19, 18, 18, 17, 17, 16, 16, 16, 16,
+      15, 15, 15, 15, 14, 14, 14, 14, 14, 14, 14, 13, 13, 13, 13, 13, 13};
+
+  static constexpr uint64_t min_safe_u64[] = {
+      9223372036854775808ull,  12157665459056928801ull, 4611686018427387904,
+      7450580596923828125,     4738381338321616896,     3909821048582988049,
+      9223372036854775808ull,  12157665459056928801ull, 10000000000000000000ull,
+      5559917313492231481,     2218611106740436992,     8650415919381337933,
+      2177953337809371136,     6568408355712890625,     1152921504606846976,
+      2862423051509815793,     6746640616477458432,     15181127029874798299ull,
+      1638400000000000000,     3243919932521508681,     6221821273427820544,
+      11592836324538749809ull, 876488338465357824,      1490116119384765625,
+      2481152873203736576,     4052555153018976267,     6502111422497947648,
+      10260628712958602189ull, 15943230000000000000ull, 787662783788549761,
+      1152921504606846976,     1667889514952984961,     2386420683693101056,
+      3379220508056640625,     4738381338321616896};
+};
+
+#if SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE
+
+template <typename T> constexpr uint8_t int_luts<T>::chdigit[];
+
+template <typename T> constexpr size_t int_luts<T>::maxdigits_u64[];
+
+template <typename T> constexpr uint64_t int_luts<T>::min_safe_u64[];
+
+#endif
+
+template <typename UC>
+simdjson_fastfloat_really_inline constexpr uint8_t ch_to_digit(UC c) {
+  // wchar_t and char can be signed, so we need to be careful.
+  using UnsignedUC = typename std::make_unsigned<UC>::type;
+  return int_luts<>::chdigit[static_cast<unsigned char>(
+      static_cast<UnsignedUC>(c) &
+      static_cast<UnsignedUC>(
+          -((static_cast<UnsignedUC>(c) & ~0xFFull) == 0)))];
+}
+
+simdjson_fastfloat_really_inline constexpr size_t max_digits_u64(int base) {
+  return int_luts<>::maxdigits_u64[base - 2];
+}
+
+// If a u64 is exactly max_digits_u64() in length, this is
+// the value below which it has definitely overflowed.
+simdjson_fastfloat_really_inline constexpr uint64_t min_safe_u64(int base) {
+  return int_luts<>::min_safe_u64[base - 2];
+}
+
+static_assert(std::is_same<equiv_uint_t<double>, uint64_t>::value,
+              "equiv_uint should be uint64_t for double");
+static_assert(std::numeric_limits<double>::is_iec559,
+              "double must fulfill the requirements of IEC 559 (IEEE 754)");
+
+static_assert(std::is_same<equiv_uint_t<float>, uint32_t>::value,
+              "equiv_uint should be uint32_t for float");
+static_assert(std::numeric_limits<float>::is_iec559,
+              "float must fulfill the requirements of IEC 559 (IEEE 754)");
+
+#ifdef __STDCPP_FLOAT64_T__
+static_assert(std::is_same<equiv_uint_t<std::float64_t>, uint64_t>::value,
+              "equiv_uint should be uint64_t for std::float64_t");
+static_assert(
+    std::numeric_limits<std::float64_t>::is_iec559,
+    "std::float64_t must fulfill the requirements of IEC 559 (IEEE 754)");
+
+template <>
+struct binary_format<std::float64_t> : public binary_format<double> {};
+#endif // __STDCPP_FLOAT64_T__
+
+#ifdef __STDCPP_FLOAT32_T__
+static_assert(std::is_same<equiv_uint_t<std::float32_t>, uint32_t>::value,
+              "equiv_uint should be uint32_t for std::float32_t");
+static_assert(
+    std::numeric_limits<std::float32_t>::is_iec559,
+    "std::float32_t must fulfill the requirements of IEC 559 (IEEE 754)");
+
+template <>
+struct binary_format<std::float32_t> : public binary_format<float> {};
+#endif // __STDCPP_FLOAT32_T__
+
+#ifdef __STDCPP_FLOAT16_T__
+static_assert(
+    std::is_same<binary_format<std::float16_t>::equiv_uint, uint16_t>::value,
+    "equiv_uint should be uint16_t for std::float16_t");
+static_assert(
+    std::numeric_limits<std::float16_t>::is_iec559,
+    "std::float16_t must fulfill the requirements of IEC 559 (IEEE 754)");
+#endif // __STDCPP_FLOAT16_T__
+
+#ifdef __STDCPP_BFLOAT16_T__
+static_assert(
+    std::is_same<binary_format<std::bfloat16_t>::equiv_uint, uint16_t>::value,
+    "equiv_uint should be uint16_t for std::bfloat16_t");
+static_assert(
+    std::numeric_limits<std::bfloat16_t>::is_iec559,
+    "std::bfloat16_t must fulfill the requirements of IEC 559 (IEEE 754)");
+#endif // __STDCPP_BFLOAT16_T__
+
+constexpr chars_format operator~(chars_format rhs) noexcept {
+  using int_type = std::underlying_type<chars_format>::type;
+  return static_cast<chars_format>(~static_cast<int_type>(rhs));
+}
+
+constexpr chars_format operator&(chars_format lhs, chars_format rhs) noexcept {
+  using int_type = std::underlying_type<chars_format>::type;
+  return static_cast<chars_format>(static_cast<int_type>(lhs) &
+                                   static_cast<int_type>(rhs));
+}
+
+constexpr chars_format operator|(chars_format lhs, chars_format rhs) noexcept {
+  using int_type = std::underlying_type<chars_format>::type;
+  return static_cast<chars_format>(static_cast<int_type>(lhs) |
+                                   static_cast<int_type>(rhs));
+}
+
+constexpr chars_format operator^(chars_format lhs, chars_format rhs) noexcept {
+  using int_type = std::underlying_type<chars_format>::type;
+  return static_cast<chars_format>(static_cast<int_type>(lhs) ^
+                                   static_cast<int_type>(rhs));
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 chars_format &
+operator&=(chars_format &lhs, chars_format rhs) noexcept {
+  return lhs = (lhs & rhs);
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 chars_format &
+operator|=(chars_format &lhs, chars_format rhs) noexcept {
+  return lhs = (lhs | rhs);
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 chars_format &
+operator^=(chars_format &lhs, chars_format rhs) noexcept {
+  return lhs = (lhs ^ rhs);
+}
+
+namespace detail {
+// adjust for deprecated feature macros
+constexpr chars_format adjust_for_feature_macros(chars_format fmt) {
+  return fmt
+#ifdef SIMDJSON_FASTFLOAT_ALLOWS_LEADING_PLUS
+         | chars_format::allow_leading_plus
+#endif
+#ifdef SIMDJSON_FASTFLOAT_SKIP_WHITE_SPACE
+         | chars_format::skip_white_space
+#endif
+      ;
+}
+} // namespace detail
+} // namespace simdjson_fast_float
+
+#endif
+
+
+#ifndef SIMDJSON_FASTFLOAT_FAST_FLOAT_H
+#define SIMDJSON_FASTFLOAT_FAST_FLOAT_H
+
+
+namespace simdjson_fast_float {
+/**
+ * This function parses the character sequence [first,last) for a number. It
+ * parses floating-point numbers expecting a locale-independent format
+ * equivalent to what is used by std::strtod in the default ("C") locale. The
+ * resulting floating-point value is the closest floating-point values (using
+ * either float or double), using the "round to even" convention for values that
+ * would otherwise fall right in-between two values. That is, we provide exact
+ * parsing according to the IEEE standard.
+ *
+ * Given a successful parse, the pointer (`ptr`) in the returned value is set to
+ * point right after the parsed number, and the `value` referenced is set to the
+ * parsed value. In case of error, the returned `ec` contains a representative
+ * error, otherwise the default (`std::errc()`) value is stored.
+ *
+ * The implementation does not throw and does not allocate memory (e.g., with
+ * `new` or `malloc`).
+ *
+ * Like the C++17 standard, the `simdjson_fast_float::from_chars` functions take an
+ * optional last argument of the type `simdjson_fast_float::chars_format`. It is a bitset
+ * value: we check whether `fmt & simdjson_fast_float::chars_format::fixed` and `fmt &
+ * simdjson_fast_float::chars_format::scientific` are set to determine whether we allow
+ * the fixed point and scientific notation respectively. The default is
+ * `simdjson_fast_float::chars_format::general` which allows both `fixed` and
+ * `scientific`.
+ */
+template <typename T, typename UC = char,
+          typename = SIMDJSON_FASTFLOAT_ENABLE_IF(is_supported_float_type<T>::value)>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+from_chars(UC const *first, UC const *last, T &value,
+           chars_format fmt = chars_format::general) noexcept;
+
+/**
+ * Like from_chars, but accepts an `options` argument to govern number parsing.
+ * Both for floating-point types and integer types.
+ */
+template <typename T, typename UC = char>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+from_chars_advanced(UC const *first, UC const *last, T &value,
+                    parse_options_t<UC> options) noexcept;
+
+/**
+ * This function multiplies an integer number by a power of 10 and returns
+ * the result as a double precision floating-point value that is correctly
+ * rounded. The resulting floating-point value is the closest floating-point
+ * value, using the "round to nearest, tie to even" convention for values that
+ * would otherwise fall right in-between two values. That is, we provide exact
+ * conversion according to the IEEE standard.
+ *
+ * On overflow infinity is returned, on underflow 0 is returned.
+ *
+ * The implementation does not throw and does not allocate memory (e.g., with
+ * `new` or `malloc`).
+ */
+SIMDJSON_FASTFLOAT_CONSTEXPR20 inline double
+integer_times_pow10(uint64_t mantissa, int decimal_exponent) noexcept;
+SIMDJSON_FASTFLOAT_CONSTEXPR20 inline double
+integer_times_pow10(int64_t mantissa, int decimal_exponent) noexcept;
+
+/**
+ * This function is a template overload of `integer_times_pow10()`
+ * that returns a floating-point value of type `T` that is one of
+ * supported floating-point types (e.g. `double`, `float`).
+ */
+template <typename T>
+SIMDJSON_FASTFLOAT_CONSTEXPR20
+    typename std::enable_if<is_supported_float_type<T>::value, T>::type
+    integer_times_pow10(uint64_t mantissa, int decimal_exponent) noexcept;
+template <typename T>
+SIMDJSON_FASTFLOAT_CONSTEXPR20
+    typename std::enable_if<is_supported_float_type<T>::value, T>::type
+    integer_times_pow10(int64_t mantissa, int decimal_exponent) noexcept;
+
+/**
+ * from_chars for integer types.
+ */
+template <typename T, typename UC = char,
+          typename = SIMDJSON_FASTFLOAT_ENABLE_IF(is_supported_integer_type<T>::value)>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+from_chars(UC const *first, UC const *last, T &value, int base = 10) noexcept;
+
+} // namespace simdjson_fast_float
+
+#endif // SIMDJSON_FASTFLOAT_FAST_FLOAT_H
+
+#ifndef SIMDJSON_FASTFLOAT_ASCII_NUMBER_H
+#define SIMDJSON_FASTFLOAT_ASCII_NUMBER_H
+
+#include <cctype>
+#include <cstdint>
+#include <cstring>
+#include <iterator>
+#include <limits>
+#include <type_traits>
+
+
+#ifdef SIMDJSON_FASTFLOAT_SSE2
+#include <emmintrin.h>
+#endif
+
+#ifdef SIMDJSON_FASTFLOAT_NEON
+#include <arm_neon.h>
+#endif
+
+namespace simdjson_fast_float {
+
+template <typename UC> simdjson_fastfloat_really_inline constexpr bool has_simd_opt() {
+#ifdef SIMDJSON_FASTFLOAT_HAS_SIMD
+  return std::is_same<UC, char16_t>::value;
+#else
+  return false;
+#endif
+}
+
+// Next function can be micro-optimized, but compilers are entirely
+// able to optimize it well.
+template <typename UC>
+simdjson_fastfloat_really_inline constexpr bool is_integer(UC c) noexcept {
+  return static_cast<unsigned>(c - UC('0')) <= 9u;
+}
+
+simdjson_fastfloat_really_inline constexpr uint64_t byteswap(uint64_t val) {
+  return (val & 0xFF00000000000000) >> 56 | (val & 0x00FF000000000000) >> 40 |
+         (val & 0x0000FF0000000000) >> 24 | (val & 0x000000FF00000000) >> 8 |
+         (val & 0x00000000FF000000) << 8 | (val & 0x0000000000FF0000) << 24 |
+         (val & 0x000000000000FF00) << 40 | (val & 0x00000000000000FF) << 56;
+}
+
+simdjson_fastfloat_really_inline constexpr uint32_t byteswap_32(uint32_t val) {
+  return (val >> 24) | ((val >> 8) & 0x0000FF00u) | ((val << 8) & 0x00FF0000u) |
+         (val << 24);
+}
+
+// Read 8 UC into a u64. Truncates UC if not char.
+template <typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 uint64_t
+read8_to_u64(UC const *chars) {
+  if (cpp20_and_in_constexpr() || !std::is_same<UC, char>::value) {
+    uint64_t val = 0;
+    for (int i = 0; i < 8; ++i) {
+      val |= uint64_t(uint8_t(*chars)) << (i * 8);
+      ++chars;
+    }
+    return val;
+  }
+  uint64_t val;
+  ::memcpy(&val, chars, sizeof(uint64_t));
+#if SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN == 1
+  // Need to read as-if the number was in little-endian order.
+  val = byteswap(val);
+#endif
+  return val;
+}
+
+// Read 4 UC into a u32. Truncates UC if not char.
+template <typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 uint32_t
+read4_to_u32(UC const *chars) {
+  if (cpp20_and_in_constexpr() || !std::is_same<UC, char>::value) {
+    uint32_t val = 0;
+    for (int i = 0; i < 4; ++i) {
+      val |= uint32_t(uint8_t(*chars)) << (i * 8);
+      ++chars;
+    }
+    return val;
+  }
+  uint32_t val;
+  ::memcpy(&val, chars, sizeof(uint32_t));
+#if SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN == 1
+  val = byteswap_32(val);
+#endif
+  return val;
+}
+#ifdef SIMDJSON_FASTFLOAT_SSE2
+
+simdjson_fastfloat_really_inline uint64_t simd_read8_to_u64(__m128i const data) {
+  SIMDJSON_FASTFLOAT_SIMD_DISABLE_WARNINGS
+  __m128i const packed = _mm_packus_epi16(data, data);
+#ifdef SIMDJSON_FASTFLOAT_64BIT
+  return uint64_t(_mm_cvtsi128_si64(packed));
+#else
+  uint64_t value;
+  // Visual Studio + older versions of GCC don't support _mm_storeu_si64
+  _mm_storel_epi64(reinterpret_cast<__m128i *>(&value), packed);
+  return value;
+#endif
+  SIMDJSON_FASTFLOAT_SIMD_RESTORE_WARNINGS
+}
+
+simdjson_fastfloat_really_inline uint64_t simd_read8_to_u64(char16_t const *chars) {
+  SIMDJSON_FASTFLOAT_SIMD_DISABLE_WARNINGS
+  return simd_read8_to_u64(
+      _mm_loadu_si128(reinterpret_cast<__m128i const *>(chars)));
+  SIMDJSON_FASTFLOAT_SIMD_RESTORE_WARNINGS
+}
+
+#elif defined(SIMDJSON_FASTFLOAT_NEON)
+
+simdjson_fastfloat_really_inline uint64_t simd_read8_to_u64(uint16x8_t const data) {
+  SIMDJSON_FASTFLOAT_SIMD_DISABLE_WARNINGS
+  uint8x8_t utf8_packed = vmovn_u16(data);
+  return vget_lane_u64(vreinterpret_u64_u8(utf8_packed), 0);
+  SIMDJSON_FASTFLOAT_SIMD_RESTORE_WARNINGS
+}
+
+simdjson_fastfloat_really_inline uint64_t simd_read8_to_u64(char16_t const *chars) {
+  SIMDJSON_FASTFLOAT_SIMD_DISABLE_WARNINGS
+  return simd_read8_to_u64(
+      vld1q_u16(reinterpret_cast<uint16_t const *>(chars)));
+  SIMDJSON_FASTFLOAT_SIMD_RESTORE_WARNINGS
+}
+
+#endif // SIMDJSON_FASTFLOAT_SSE2
+
+// MSVC SFINAE is broken pre-VS2017
+#if defined(_MSC_VER) && _MSC_VER <= 1900
+template <typename UC>
+#else
+template <typename UC, SIMDJSON_FASTFLOAT_ENABLE_IF(!has_simd_opt<UC>()) = 0>
+#endif
+// dummy for compile
+uint64_t simd_read8_to_u64(UC const *) {
+  return 0;
+}
+
+// credit  @aqrit
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 uint32_t
+parse_eight_digits_unrolled(uint64_t val) {
+  uint64_t const mask = 0x000000FF000000FF;
+  uint64_t const mul1 = 0x000F424000000064; // 100 + (1000000ULL << 32)
+  uint64_t const mul2 = 0x0000271000000001; // 1 + (10000ULL << 32)
+  val -= 0x3030303030303030;
+  val = (val * 10) + (val >> 8); // val = (val * 2561) >> 8;
+  val = (((val & mask) * mul1) + (((val >> 16) & mask) * mul2)) >> 32;
+  return uint32_t(val);
+}
+
+// Call this if chars are definitely 8 digits.
+template <typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 uint32_t
+parse_eight_digits_unrolled(UC const *chars) noexcept {
+  if (cpp20_and_in_constexpr() || !has_simd_opt<UC>()) {
+    return parse_eight_digits_unrolled(read8_to_u64(chars)); // truncation okay
+  }
+  return parse_eight_digits_unrolled(simd_read8_to_u64(chars));
+}
+
+// credit @aqrit
+simdjson_fastfloat_really_inline constexpr bool
+is_made_of_eight_digits_fast(uint64_t val) noexcept {
+  return !((((val + 0x4646464646464646) | (val - 0x3030303030303030)) &
+            0x8080808080808080));
+}
+
+simdjson_fastfloat_really_inline constexpr bool
+is_made_of_four_digits_fast(uint32_t val) noexcept {
+  return !((((val + 0x46464646) | (val - 0x30303030)) & 0x80808080));
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 uint32_t
+parse_four_digits_unrolled(uint32_t val) noexcept {
+  val -= 0x30303030;
+  val = (val * 10) + (val >> 8);
+  return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
+#ifdef SIMDJSON_FASTFLOAT_HAS_SIMD
+
+// Call this if chars might not be 8 digits.
+// Using this style (instead of is_made_of_eight_digits_fast() then
+// parse_eight_digits_unrolled()) ensures we don't load SIMD registers twice.
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 bool
+simd_parse_if_eight_digits_unrolled(char16_t const *chars,
+                                    uint64_t &i) noexcept {
+  if (cpp20_and_in_constexpr()) {
+    return false;
+  }
+#ifdef SIMDJSON_FASTFLOAT_SSE2
+  SIMDJSON_FASTFLOAT_SIMD_DISABLE_WARNINGS
+  __m128i const data =
+      _mm_loadu_si128(reinterpret_cast<__m128i const *>(chars));
+
+  // (x - '0') <= 9
+  // http://0x80.pl/articles/simd-parsing-int-sequences.html
+  __m128i const t0 = _mm_add_epi16(data, _mm_set1_epi16(32720));
+  __m128i const t1 = _mm_cmpgt_epi16(t0, _mm_set1_epi16(-32759));
+
+  if (_mm_movemask_epi8(t1) == 0) {
+    i = i * 100000000 + parse_eight_digits_unrolled(simd_read8_to_u64(data));
+    return true;
+  } else
+    return false;
+  SIMDJSON_FASTFLOAT_SIMD_RESTORE_WARNINGS
+#elif defined(SIMDJSON_FASTFLOAT_NEON)
+  SIMDJSON_FASTFLOAT_SIMD_DISABLE_WARNINGS
+  uint16x8_t const data = vld1q_u16(reinterpret_cast<uint16_t const *>(chars));
+
+  // (x - '0') <= 9
+  // http://0x80.pl/articles/simd-parsing-int-sequences.html
+  uint16x8_t const t0 = vsubq_u16(data, vmovq_n_u16('0'));
+  uint16x8_t const mask = vcltq_u16(t0, vmovq_n_u16('9' - '0' + 1));
+
+  if (vminvq_u16(mask) == 0xFFFF) {
+    i = i * 100000000 + parse_eight_digits_unrolled(simd_read8_to_u64(data));
+    return true;
+  } else
+    return false;
+  SIMDJSON_FASTFLOAT_SIMD_RESTORE_WARNINGS
+#else
+  static_cast<void>(chars);
+  static_cast<void>(i);
+  return false;
+#endif // SIMDJSON_FASTFLOAT_SSE2
+}
+
+#endif // SIMDJSON_FASTFLOAT_HAS_SIMD
+
+// MSVC SFINAE is broken pre-VS2017
+#if defined(_MSC_VER) && _MSC_VER <= 1900
+template <typename UC>
+#else
+template <typename UC, SIMDJSON_FASTFLOAT_ENABLE_IF(!has_simd_opt<UC>()) = 0>
+#endif
+// dummy for compile
+bool simd_parse_if_eight_digits_unrolled(UC const *, uint64_t &) {
+  return 0;
+}
+
+template <typename UC, SIMDJSON_FASTFLOAT_ENABLE_IF(!std::is_same<UC, char>::value) = 0>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 void
+loop_parse_if_eight_digits(UC const *&p, UC const *const pend, uint64_t &i) {
+  if (!has_simd_opt<UC>()) {
+    return;
+  }
+  while ((std::distance(p, pend) >= 8) &&
+         simd_parse_if_eight_digits_unrolled(
+             p, i)) { // in rare cases, this will overflow, but that's ok
+    p += 8;
+  }
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 void
+loop_parse_if_eight_digits(char const *&p, char const *const pend,
+                           uint64_t &i) {
+  // optimizes better than parse_if_eight_digits_unrolled() for UC = char.
+  while ((std::distance(p, pend) >= 8) &&
+         is_made_of_eight_digits_fast(read8_to_u64(p))) {
+    i = i * 100000000 +
+        parse_eight_digits_unrolled(read8_to_u64(
+            p)); // in rare cases, this will overflow, but that's ok
+    p += 8;
+  }
+  // Consume a remaining 4-7 digit run in a single SWAR step instead of
+  // byte-by-byte (reuses the existing 4-digit helpers). The parsed result is
+  // identical either way. Historically gated to clang because gcc regressed on
+  // short remainders, but that verdict predates the span-elision restructure;
+  // with the leaner hot path the 4-digit step now wins on gcc as well.
+  if ((pend - p) >= 4) {
+    uint32_t const val4 = read4_to_u32(p);
+    if (is_made_of_four_digits_fast(val4)) {
+      i = i * 10000 +
+          parse_four_digits_unrolled(val4); // may overflow, that's ok
+      p += 4;
+    }
+  }
+}
+
+enum class parse_error {
+  no_error,
+  // [JSON-only] The minus sign must be followed by an integer.
+  missing_integer_after_sign,
+  // A sign must be followed by an integer or dot.
+  missing_integer_or_dot_after_sign,
+  // [JSON-only] The integer part must not have leading zeros.
+  leading_zeros_in_integer_part,
+  // [JSON-only] The integer part must have at least one digit.
+  no_digits_in_integer_part,
+  // [JSON-only] If there is a decimal point, there must be digits in the
+  // fractional part.
+  no_digits_in_fractional_part,
+  // The mantissa must have at least one digit.
+  no_digits_in_mantissa,
+  // Scientific notation requires an exponential part.
+  missing_exponential_part,
+};
+
+template <typename UC> struct parsed_number_string_t {
+  int64_t exponent{0};
+  uint64_t mantissa{0};
+  UC const *lastmatch{nullptr};
+  bool negative{false};
+  bool valid{false};
+  bool too_many_digits{false};
+  // contains the range of the significant digits
+  span<UC const> integer{};  // non-nullable
+  span<UC const> fraction{}; // nullable
+  parse_error error{parse_error::no_error};
+};
+
+using byte_span = span<char const>;
+using parsed_number_string = parsed_number_string_t<char>;
+
+template <typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 parsed_number_string_t<UC>
+report_parse_error(UC const *p, parse_error error) {
+  parsed_number_string_t<UC> answer;
+  answer.valid = false;
+  answer.lastmatch = p;
+  answer.error = error;
+  return answer;
+}
+
+// Assuming that you use no more than 19 digits, this will
+// parse an ASCII string.
+//
+// store_spans is a *runtime* flag (not a template parameter, deliberately: a
+// template would create a second instantiation of this whole function and the
+// extra icache pressure wipes out the gain). When false, the integer/fraction
+// spans (read only by the rare digit_comp slow path) are not materialized,
+// which keeps the fat parsed_number_string_t off the hot path. The caller
+// re-parses with store_spans=true if the slow path is actually reached.
+template <bool basic_json_fmt, typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 parsed_number_string_t<UC>
+parse_number_string(UC const *p, UC const *pend, parse_options_t<UC> options,
+                    bool store_spans = true) noexcept {
+  chars_format const fmt = detail::adjust_for_feature_macros(options.format);
+  UC const decimal_point = options.decimal_point;
+
+  parsed_number_string_t<UC> answer;
+  answer.valid = false;
+  answer.too_many_digits = false;
+  // assume p < pend, so dereference without checks;
+  answer.negative = (*p == UC('-'));
+  // C++17 20.19.3.(7.1) explicitly forbids '+' sign here
+  if ((*p == UC('-')) || (uint64_t(fmt & chars_format::allow_leading_plus) &&
+                          !basic_json_fmt && *p == UC('+'))) {
+    ++p;
+    if (p == pend) {
+      return report_parse_error<UC>(
+          p, parse_error::missing_integer_or_dot_after_sign);
+    }
+    SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(basic_json_fmt) {
+      if (!is_integer(*p)) { // a sign must be followed by an integer
+        return report_parse_error<UC>(p,
+                                      parse_error::missing_integer_after_sign);
+      }
+    }
+    else {
+      if (!is_integer(*p) &&
+          (*p !=
+           decimal_point)) { // a sign must be followed by an integer or the dot
+        return report_parse_error<UC>(
+            p, parse_error::missing_integer_or_dot_after_sign);
+      }
+    }
+  }
+  UC const *const start_digits = p;
+
+  uint64_t i = 0; // an unsigned int avoids signed overflows (which are bad)
+
+  // Straight-line unroll of the integer-part scan: most integer parts are
+  // 1-5 digits, so peeling the first iterations eliminates the loop back-edge
+  // for the common case. Semantics are identical to the original `while` loop:
+  // i = 10*i + digit, advancing p.
+  if ((p != pend) && is_integer(*p)) {
+    i = uint64_t(*p - UC('0'));
+    ++p;
+    if ((p != pend) && is_integer(*p)) {
+      i = 10 * i + uint64_t(*p - UC('0'));
+      ++p;
+      if ((p != pend) && is_integer(*p)) {
+        i = 10 * i + uint64_t(*p - UC('0'));
+        ++p;
+        if ((p != pend) && is_integer(*p)) {
+          i = 10 * i + uint64_t(*p - UC('0'));
+          ++p;
+          if ((p != pend) && is_integer(*p)) {
+            i = 10 * i + uint64_t(*p - UC('0'));
+            ++p;
+            while ((p != pend) && is_integer(*p)) {
+              // a multiplication by 10 is cheaper than an arbitrary integer
+              // multiplication
+              i = 10 * i +
+                  uint64_t(*p - UC('0')); // might overflow, handled later
+              ++p;
+            }
+          }
+        }
+      }
+    }
+  }
+  UC const *const end_of_integer_part = p;
+  int64_t digit_count = int64_t(end_of_integer_part - start_digits);
+  if (store_spans) {
+    answer.integer = span<UC const>(start_digits, size_t(digit_count));
+  }
+  SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(basic_json_fmt) {
+    // at least 1 digit in integer part, without leading zeros
+    if (digit_count == 0) {
+      return report_parse_error<UC>(p, parse_error::no_digits_in_integer_part);
+    }
+    if ((start_digits[0] == UC('0') && digit_count > 1)) {
+      return report_parse_error<UC>(start_digits,
+                                    parse_error::leading_zeros_in_integer_part);
+    }
+  }
+
+  int64_t exponent = 0;
+  bool const has_decimal_point = (p != pend) && (*p == decimal_point);
+  if (has_decimal_point) {
+    ++p;
+    UC const *before = p;
+    // can occur at most twice without overflowing, but let it occur more, since
+    // for integers with many digits, digit parsing is the primary bottleneck.
+    loop_parse_if_eight_digits(p, pend, i);
+
+    while ((p != pend) && is_integer(*p)) {
+      uint8_t digit = uint8_t(*p - UC('0'));
+      ++p;
+      i = i * 10 + digit; // in rare cases, this will overflow, but that's ok
+    }
+    exponent = before - p;
+    if (store_spans) {
+      answer.fraction = span<UC const>(before, size_t(p - before));
+    }
+    digit_count -= exponent;
+  }
+  SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(basic_json_fmt) {
+    // at least 1 digit in fractional part
+    if (has_decimal_point && exponent == 0) {
+      return report_parse_error<UC>(p,
+                                    parse_error::no_digits_in_fractional_part);
+    }
+  }
+  else if (digit_count == 0) { // we must have encountered at least one integer!
+    return report_parse_error<UC>(p, parse_error::no_digits_in_mantissa);
+  }
+  int64_t exp_number = 0; // explicit exponential part
+  if ((uint64_t(fmt & chars_format::scientific) && (p != pend) &&
+       ((UC('e') == *p) || (UC('E') == *p))) ||
+      (uint64_t(fmt & detail::basic_fortran_fmt) && (p != pend) &&
+       ((UC('+') == *p) || (UC('-') == *p) || (UC('d') == *p) ||
+        (UC('D') == *p)))) {
+    UC const *location_of_e = p;
+    if ((UC('e') == *p) || (UC('E') == *p) || (UC('d') == *p) ||
+        (UC('D') == *p)) {
+      ++p;
+    }
+    bool neg_exp = false;
+    if ((p != pend) && (UC('-') == *p)) {
+      neg_exp = true;
+      ++p;
+    } else if ((p != pend) &&
+               (UC('+') ==
+                *p)) { // '+' on exponent is allowed by C++17 20.19.3.(7.1)
+      ++p;
+    }
+    if ((p == pend) || !is_integer(*p)) {
+      if (!uint64_t(fmt & chars_format::fixed)) {
+        // The exponential part is invalid for scientific notation, so it must
+        // be a trailing token for fixed notation. However, fixed notation is
+        // disabled, so report a scientific notation error.
+        return report_parse_error<UC>(p, parse_error::missing_exponential_part);
+      }
+      // Otherwise, we will be ignoring the 'e'.
+      p = location_of_e;
+    } else {
+      while ((p != pend) && is_integer(*p)) {
+        uint8_t digit = uint8_t(*p - UC('0'));
+        if (exp_number < 0x10000000) {
+          exp_number = 10 * exp_number + digit;
+        }
+        ++p;
+      }
+      if (neg_exp) {
+        exp_number = -exp_number;
+      }
+      exponent += exp_number;
+    }
+  } else {
+    // If it scientific and not fixed, we have to bail out.
+    if (uint64_t(fmt & chars_format::scientific) &&
+        !uint64_t(fmt & chars_format::fixed)) {
+      return report_parse_error<UC>(p, parse_error::missing_exponential_part);
+    }
+  }
+  answer.lastmatch = p;
+  answer.valid = true;
+
+  // If we frequently had to deal with long strings of digits,
+  // we could extend our code by using a 128-bit integer instead
+  // of a 64-bit integer. However, this is uncommon.
+  //
+  // We can deal with up to 19 digits.
+  if (digit_count > 19) { // this is uncommon
+    // It is possible that the integer had an overflow.
+    // We have to handle the case where we have 0.0000somenumber.
+    // We need to be mindful of the case where we only have zeroes...
+    // E.g., 0.000000000...000.
+    UC const *start = start_digits;
+    while ((start != pend) && (*start == UC('0') || *start == decimal_point)) {
+      if (*start == UC('0')) {
+        digit_count--;
+      }
+      start++;
+    }
+
+    if (digit_count > 19) {
+      answer.too_many_digits = true;
+      // The truncation recompute below reads the integer/fraction spans. When
+      // store_spans is false we didn't materialize them, so just flag
+      // too_many_digits; the caller re-parses with store_spans=true to obtain
+      // the corrected mantissa/exponent before taking the slow path.
+      if (store_spans) {
+        // Let us start again, this time, avoiding overflows.
+        // We don't need to call if is_integer, since we use the
+        // pre-tokenized spans from above.
+        i = 0;
+        p = answer.integer.ptr;
+        UC const *int_end = p + answer.integer.len();
+        uint64_t const minimal_nineteen_digit_integer{1000000000000000000};
+        while ((i < minimal_nineteen_digit_integer) && (p != int_end)) {
+          i = i * 10 + uint64_t(*p - UC('0'));
+          ++p;
+        }
+        if (i >= minimal_nineteen_digit_integer) { // We have a big integer
+          exponent = end_of_integer_part - p + exp_number;
+        } else { // We have a value with a fractional component.
+          p = answer.fraction.ptr;
+          UC const *frac_end = p + answer.fraction.len();
+          while ((i < minimal_nineteen_digit_integer) && (p != frac_end)) {
+            i = i * 10 + uint64_t(*p - UC('0'));
+            ++p;
+          }
+          exponent = answer.fraction.ptr - p + exp_number;
+        }
+        // We have now corrected both exponent and i, to a truncated value
+      }
+    }
+  }
+  answer.exponent = exponent;
+  answer.mantissa = i;
+  return answer;
+}
+
+template <typename T, typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+parse_int_string(UC const *p, UC const *pend, T &value,
+                 parse_options_t<UC> options) {
+  chars_format const fmt = detail::adjust_for_feature_macros(options.format);
+  int const base = options.base;
+
+  from_chars_result_t<UC> answer;
+
+  UC const *const first = p;
+
+  bool const negative = (*p == UC('-'));
+#ifdef SIMDJSON_FASTFLOAT_VISUAL_STUDIO
+#pragma warning(push)
+#pragma warning(disable : 4127)
+#endif
+  if (!std::is_signed<T>::value && negative) {
+#ifdef SIMDJSON_FASTFLOAT_VISUAL_STUDIO
+#pragma warning(pop)
+#endif
+    answer.ec = std::errc::invalid_argument;
+    answer.ptr = first;
+    return answer;
+  }
+  if ((*p == UC('-')) ||
+      (uint64_t(fmt & chars_format::allow_leading_plus) && (*p == UC('+')))) {
+    ++p;
+  }
+
+  UC const *const start_num = p;
+
+  while (p != pend && *p == UC('0')) {
+    ++p;
+  }
+
+  bool const has_leading_zeros = p > start_num;
+
+  UC const *const start_digits = p;
+
+  SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(
+      (std::is_same<T, std::uint8_t>::value && sizeof(UC) == 1)) {
+    if (base == 10) {
+      const size_t len = static_cast<size_t>(pend - p);
+      if (len == 0) {
+        if (has_leading_zeros) {
+          value = 0;
+          answer.ec = std::errc();
+          answer.ptr = p;
+        } else {
+          answer.ec = std::errc::invalid_argument;
+          answer.ptr = first;
+        }
+        return answer;
+      }
+
+      uint32_t digits;
+
+#if SIMDJSON_FASTFLOAT_HAS_IS_CONSTANT_EVALUATED && SIMDJSON_FASTFLOAT_HAS_BIT_CAST
+      if (std::is_constant_evaluated()) {
+        uint8_t str[4]{};
+        for (size_t j = 0; j < 4 && j < len; ++j) {
+          str[j] = static_cast<uint8_t>(p[j]);
+        }
+        digits = std::bit_cast<uint32_t>(str);
+#if SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN
+        digits = byteswap_32(digits);
+#endif
+      }
+#else
+      if (false) {
+      }
+#endif
+      else if (len >= 4) {
+        ::memcpy(&digits, p, 4);
+#if SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN
+        digits = byteswap_32(digits);
+#endif
+      } else {
+        uint32_t b0 = static_cast<uint8_t>(p[0]);
+        uint32_t b1 = (len > 1) ? static_cast<uint8_t>(p[1]) : 0xFFu;
+        uint32_t b2 = (len > 2) ? static_cast<uint8_t>(p[2]) : 0xFFu;
+        uint32_t b3 = 0xFFu;
+        digits = b0 | (b1 << 8) | (b2 << 16) | (b3 << 24);
+      }
+
+      uint32_t magic =
+          ((digits + 0x46464646u) | (digits - 0x30303030u)) & 0x80808080u;
+      uint32_t tz =
+          static_cast<uint32_t>(countr_zero_32(magic)); // 7, 15, 23, 31, or 32
+      uint32_t nd = (tz == 32) ? 4 : (tz >> 3);
+      nd = static_cast<uint32_t>(nd < len ? nd : len);
+      if (nd == 0) {
+        if (has_leading_zeros) {
+          value = 0;
+          answer.ec = std::errc();
+          answer.ptr = p;
+          return answer;
+        }
+        answer.ec = std::errc::invalid_argument;
+        answer.ptr = first;
+        return answer;
+      }
+      if (nd > 3) {
+        const UC *q = p + nd;
+        size_t rem = len - nd;
+        while (rem) {
+          if (*q < UC('0') || *q > UC('9'))
+            break;
+          ++q;
+          --rem;
+        }
+        answer.ec = std::errc::result_out_of_range;
+        answer.ptr = q;
+        return answer;
+      }
+
+      digits ^= 0x30303030u;
+      digits <<= ((4 - nd) * 8);
+
+      uint32_t check = ((digits >> 24) & 0xff) | ((digits >> 8) & 0xff00) |
+                       ((digits << 8) & 0xff0000);
+      if (check > 0x00020505) {
+        answer.ec = std::errc::result_out_of_range;
+        answer.ptr = p + nd;
+        return answer;
+      }
+      value = static_cast<uint8_t>((0x640a01 * digits) >> 24);
+      answer.ec = std::errc();
+      answer.ptr = p + nd;
+      return answer;
+    }
+  }
+
+  SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(
+      (std::is_same<T, std::uint16_t>::value && sizeof(UC) == 1)) {
+    if (base == 10) {
+      const size_t len = size_t(pend - p);
+      if (len == 0) {
+        if (has_leading_zeros) {
+          value = 0;
+          answer.ec = std::errc();
+          answer.ptr = p;
+        } else {
+          answer.ec = std::errc::invalid_argument;
+          answer.ptr = first;
+        }
+        return answer;
+      }
+
+      if (len >= 4) {
+        uint32_t digits = read4_to_u32(p);
+        if (is_made_of_four_digits_fast(digits)) {
+          uint32_t v = parse_four_digits_unrolled(digits);
+          if (len >= 5 && is_integer(p[4])) {
+            v = v * 10 + uint32_t(p[4] - '0');
+            if (len >= 6 && is_integer(p[5])) {
+              answer.ec = std::errc::result_out_of_range;
+              const UC *q = p + 5;
+              while (q != pend && is_integer(*q)) {
+                q++;
+              }
+              answer.ptr = q;
+              return answer;
+            }
+            if (v > 65535) {
+              answer.ec = std::errc::result_out_of_range;
+              answer.ptr = p + 5;
+              return answer;
+            }
+            value = uint16_t(v);
+            answer.ec = std::errc();
+            answer.ptr = p + 5;
+            return answer;
+          }
+          // 4 digits
+          value = uint16_t(v);
+          answer.ec = std::errc();
+          answer.ptr = p + 4;
+          return answer;
+        }
+      }
+    }
+  }
+
+  uint64_t i = 0;
+  if (base == 10) {
+    loop_parse_if_eight_digits(p, pend, i); // use SIMD if possible
+  }
+  while (p != pend) {
+    uint8_t digit = ch_to_digit(*p);
+    if (digit >= base) {
+      break;
+    }
+    i = uint64_t(base) * i + digit; // might overflow, check this later
+    p++;
+  }
+
+  size_t digit_count = size_t(p - start_digits);
+
+  if (digit_count == 0) {
+    if (has_leading_zeros) {
+      value = 0;
+      answer.ec = std::errc();
+      answer.ptr = p;
+    } else {
+      answer.ec = std::errc::invalid_argument;
+      answer.ptr = first;
+    }
+    return answer;
+  }
+
+  answer.ptr = p;
+
+  // check u64 overflow
+  size_t max_digits = max_digits_u64(base);
+  if (digit_count > max_digits) {
+    answer.ec = std::errc::result_out_of_range;
+    return answer;
+  }
+  // this check can be eliminated for all other types, but they will all require
+  // a max_digits(base) equivalent
+  if (digit_count == max_digits) {
+    // At the max_digits boundary the accumulator `i` may have wrapped around
+    // 2^64. A plain `i < min_safe_u64(base)` test is not sufficient: for any
+    // base whose max_digits-length range exceeds 2^64 (base 10 reaches
+    // ~5.4 * 2^64 at 20 digits) the value can wrap a whole multiple of 2^64 and
+    // land back above min_safe, slipping through. Decide exactly in O(1) using
+    // the leading digit, following the approach used in simdjson:
+    //   ms   == min_safe_u64(base) == base^(max_digits-1), the smallest
+    //           max_digits-length value.
+    //   dmax == the largest leading digit whose number can still fit in u64.
+    // The leading-digit band [d*ms, (d+1)*ms) has width ms < 2^64, so within
+    // the single band where d == dmax the value straddles 2^64 at most once,
+    // and a single threshold separates wrapped from non-wrapped values. A
+    // leading digit above dmax always overflows; below dmax always fits.
+    uint64_t const ms = min_safe_u64(base);
+    uint64_t const dmax = (std::numeric_limits<uint64_t>::max)() / ms;
+    uint64_t const lead = ch_to_digit(*start_digits);
+    if (lead > dmax || (lead == dmax && i < dmax * ms)) {
+      answer.ec = std::errc::result_out_of_range;
+      return answer;
+    }
+  }
+
+  // check other types overflow
+  if (!std::is_same<T, uint64_t>::value) {
+    if (i > uint64_t((std::numeric_limits<T>::max)()) + uint64_t(negative)) {
+      answer.ec = std::errc::result_out_of_range;
+      return answer;
+    }
+  }
+
+  if (negative) {
+#ifdef SIMDJSON_FASTFLOAT_VISUAL_STUDIO
+#pragma warning(push)
+#pragma warning(disable : 4146)
+#endif
+    // this weird workaround is required because:
+    // - converting unsigned to signed when its value is greater than signed max
+    // is UB pre-C++23.
+    // - reinterpret_casting (~i + 1) would work, but it is not constexpr
+    // this is always optimized into a neg instruction (note: T is an integer
+    // type)
+    value = T(-(std::numeric_limits<T>::max)() -
+              T(i - uint64_t((std::numeric_limits<T>::max)())));
+#ifdef SIMDJSON_FASTFLOAT_VISUAL_STUDIO
+#pragma warning(pop)
+#endif
+  } else {
+    value = T(i);
+  }
+
+  answer.ec = std::errc();
+  return answer;
+}
+
+} // namespace simdjson_fast_float
+
+#endif
+
+#ifndef SIMDJSON_FASTFLOAT_FAST_TABLE_H
+#define SIMDJSON_FASTFLOAT_FAST_TABLE_H
+
+#include <cstdint>
+
+namespace simdjson_fast_float {
+
+/**
+ * When mapping numbers from decimal to binary,
+ * we go from w * 10^q to m * 2^p but we have
+ * 10^q = 5^q * 2^q, so effectively
+ * we are trying to match
+ * w * 2^q * 5^q to m * 2^p. Thus the powers of two
+ * are not a concern since they can be represented
+ * exactly using the binary notation, only the powers of five
+ * affect the binary significand.
+ */
+
+/**
+ * The smallest non-zero float (binary64) is 2^-1074.
+ * We take as input numbers of the form w x 10^q where w < 2^64.
+ * We have that w * 10^-343  <  2^(64-344) 5^-343 < 2^-1076.
+ * However, we have that
+ * (2^64-1) * 10^-342 =  (2^64-1) * 2^-342 * 5^-342 > 2^-1074.
+ * Thus it is possible for a number of the form w * 10^-342 where
+ * w is a 64-bit value to be a non-zero floating-point number.
+ *********
+ * Any number of form w * 10^309 where w>= 1 is going to be
+ * infinite in binary64 so we never need to worry about powers
+ * of 5 greater than 308.
+ */
+template <class unused = void> struct powers_template {
+
+  constexpr static int smallest_power_of_five =
+      binary_format<double>::smallest_power_of_ten();
+  constexpr static int largest_power_of_five =
+      binary_format<double>::largest_power_of_ten();
+  constexpr static int number_of_entries =
+      2 * (largest_power_of_five - smallest_power_of_five + 1);
+  // Powers of five from 5^-342 all the way to 5^308 rounded toward one.
+  constexpr static uint64_t power_of_five_128[number_of_entries] = {
+      0xeef453d6923bd65a, 0x113faa2906a13b3f,
+      0x9558b4661b6565f8, 0x4ac7ca59a424c507,
+      0xbaaee17fa23ebf76, 0x5d79bcf00d2df649,
+      0xe95a99df8ace6f53, 0xf4d82c2c107973dc,
+      0x91d8a02bb6c10594, 0x79071b9b8a4be869,
+      0xb64ec836a47146f9, 0x9748e2826cdee284,
+      0xe3e27a444d8d98b7, 0xfd1b1b2308169b25,
+      0x8e6d8c6ab0787f72, 0xfe30f0f5e50e20f7,
+      0xb208ef855c969f4f, 0xbdbd2d335e51a935,
+      0xde8b2b66b3bc4723, 0xad2c788035e61382,
+      0x8b16fb203055ac76, 0x4c3bcb5021afcc31,
+      0xaddcb9e83c6b1793, 0xdf4abe242a1bbf3d,
+      0xd953e8624b85dd78, 0xd71d6dad34a2af0d,
+      0x87d4713d6f33aa6b, 0x8672648c40e5ad68,
+      0xa9c98d8ccb009506, 0x680efdaf511f18c2,
+      0xd43bf0effdc0ba48, 0x212bd1b2566def2,
+      0x84a57695fe98746d, 0x14bb630f7604b57,
+      0xa5ced43b7e3e9188, 0x419ea3bd35385e2d,
+      0xcf42894a5dce35ea, 0x52064cac828675b9,
+      0x818995ce7aa0e1b2, 0x7343efebd1940993,
+      0xa1ebfb4219491a1f, 0x1014ebe6c5f90bf8,
+      0xca66fa129f9b60a6, 0xd41a26e077774ef6,
+      0xfd00b897478238d0, 0x8920b098955522b4,
+      0x9e20735e8cb16382, 0x55b46e5f5d5535b0,
+      0xc5a890362fddbc62, 0xeb2189f734aa831d,
+      0xf712b443bbd52b7b, 0xa5e9ec7501d523e4,
+      0x9a6bb0aa55653b2d, 0x47b233c92125366e,
+      0xc1069cd4eabe89f8, 0x999ec0bb696e840a,
+      0xf148440a256e2c76, 0xc00670ea43ca250d,
+      0x96cd2a865764dbca, 0x380406926a5e5728,
+      0xbc807527ed3e12bc, 0xc605083704f5ecf2,
+      0xeba09271e88d976b, 0xf7864a44c633682e,
+      0x93445b8731587ea3, 0x7ab3ee6afbe0211d,
+      0xb8157268fdae9e4c, 0x5960ea05bad82964,
+      0xe61acf033d1a45df, 0x6fb92487298e33bd,
+      0x8fd0c16206306bab, 0xa5d3b6d479f8e056,
+      0xb3c4f1ba87bc8696, 0x8f48a4899877186c,
+      0xe0b62e2929aba83c, 0x331acdabfe94de87,
+      0x8c71dcd9ba0b4925, 0x9ff0c08b7f1d0b14,
+      0xaf8e5410288e1b6f, 0x7ecf0ae5ee44dd9,
+      0xdb71e91432b1a24a, 0xc9e82cd9f69d6150,
+      0x892731ac9faf056e, 0xbe311c083a225cd2,
+      0xab70fe17c79ac6ca, 0x6dbd630a48aaf406,
+      0xd64d3d9db981787d, 0x92cbbccdad5b108,
+      0x85f0468293f0eb4e, 0x25bbf56008c58ea5,
+      0xa76c582338ed2621, 0xaf2af2b80af6f24e,
+      0xd1476e2c07286faa, 0x1af5af660db4aee1,
+      0x82cca4db847945ca, 0x50d98d9fc890ed4d,
+      0xa37fce126597973c, 0xe50ff107bab528a0,
+      0xcc5fc196fefd7d0c, 0x1e53ed49a96272c8,
+      0xff77b1fcbebcdc4f, 0x25e8e89c13bb0f7a,
+      0x9faacf3df73609b1, 0x77b191618c54e9ac,
+      0xc795830d75038c1d, 0xd59df5b9ef6a2417,
+      0xf97ae3d0d2446f25, 0x4b0573286b44ad1d,
+      0x9becce62836ac577, 0x4ee367f9430aec32,
+      0xc2e801fb244576d5, 0x229c41f793cda73f,
+      0xf3a20279ed56d48a, 0x6b43527578c1110f,
+      0x9845418c345644d6, 0x830a13896b78aaa9,
+      0xbe5691ef416bd60c, 0x23cc986bc656d553,
+      0xedec366b11c6cb8f, 0x2cbfbe86b7ec8aa8,
+      0x94b3a202eb1c3f39, 0x7bf7d71432f3d6a9,
+      0xb9e08a83a5e34f07, 0xdaf5ccd93fb0cc53,
+      0xe858ad248f5c22c9, 0xd1b3400f8f9cff68,
+      0x91376c36d99995be, 0x23100809b9c21fa1,
+      0xb58547448ffffb2d, 0xabd40a0c2832a78a,
+      0xe2e69915b3fff9f9, 0x16c90c8f323f516c,
+      0x8dd01fad907ffc3b, 0xae3da7d97f6792e3,
+      0xb1442798f49ffb4a, 0x99cd11cfdf41779c,
+      0xdd95317f31c7fa1d, 0x40405643d711d583,
+      0x8a7d3eef7f1cfc52, 0x482835ea666b2572,
+      0xad1c8eab5ee43b66, 0xda3243650005eecf,
+      0xd863b256369d4a40, 0x90bed43e40076a82,
+      0x873e4f75e2224e68, 0x5a7744a6e804a291,
+      0xa90de3535aaae202, 0x711515d0a205cb36,
+      0xd3515c2831559a83, 0xd5a5b44ca873e03,
+      0x8412d9991ed58091, 0xe858790afe9486c2,
+      0xa5178fff668ae0b6, 0x626e974dbe39a872,
+      0xce5d73ff402d98e3, 0xfb0a3d212dc8128f,
+      0x80fa687f881c7f8e, 0x7ce66634bc9d0b99,
+      0xa139029f6a239f72, 0x1c1fffc1ebc44e80,
+      0xc987434744ac874e, 0xa327ffb266b56220,
+      0xfbe9141915d7a922, 0x4bf1ff9f0062baa8,
+      0x9d71ac8fada6c9b5, 0x6f773fc3603db4a9,
+      0xc4ce17b399107c22, 0xcb550fb4384d21d3,
+      0xf6019da07f549b2b, 0x7e2a53a146606a48,
+      0x99c102844f94e0fb, 0x2eda7444cbfc426d,
+      0xc0314325637a1939, 0xfa911155fefb5308,
+      0xf03d93eebc589f88, 0x793555ab7eba27ca,
+      0x96267c7535b763b5, 0x4bc1558b2f3458de,
+      0xbbb01b9283253ca2, 0x9eb1aaedfb016f16,
+      0xea9c227723ee8bcb, 0x465e15a979c1cadc,
+      0x92a1958a7675175f, 0xbfacd89ec191ec9,
+      0xb749faed14125d36, 0xcef980ec671f667b,
+      0xe51c79a85916f484, 0x82b7e12780e7401a,
+      0x8f31cc0937ae58d2, 0xd1b2ecb8b0908810,
+      0xb2fe3f0b8599ef07, 0x861fa7e6dcb4aa15,
+      0xdfbdcece67006ac9, 0x67a791e093e1d49a,
+      0x8bd6a141006042bd, 0xe0c8bb2c5c6d24e0,
+      0xaecc49914078536d, 0x58fae9f773886e18,
+      0xda7f5bf590966848, 0xaf39a475506a899e,
+      0x888f99797a5e012d, 0x6d8406c952429603,
+      0xaab37fd7d8f58178, 0xc8e5087ba6d33b83,
+      0xd5605fcdcf32e1d6, 0xfb1e4a9a90880a64,
+      0x855c3be0a17fcd26, 0x5cf2eea09a55067f,
+      0xa6b34ad8c9dfc06f, 0xf42faa48c0ea481e,
+      0xd0601d8efc57b08b, 0xf13b94daf124da26,
+      0x823c12795db6ce57, 0x76c53d08d6b70858,
+      0xa2cb1717b52481ed, 0x54768c4b0c64ca6e,
+      0xcb7ddcdda26da268, 0xa9942f5dcf7dfd09,
+      0xfe5d54150b090b02, 0xd3f93b35435d7c4c,
+      0x9efa548d26e5a6e1, 0xc47bc5014a1a6daf,
+      0xc6b8e9b0709f109a, 0x359ab6419ca1091b,
+      0xf867241c8cc6d4c0, 0xc30163d203c94b62,
+      0x9b407691d7fc44f8, 0x79e0de63425dcf1d,
+      0xc21094364dfb5636, 0x985915fc12f542e4,
+      0xf294b943e17a2bc4, 0x3e6f5b7b17b2939d,
+      0x979cf3ca6cec5b5a, 0xa705992ceecf9c42,
+      0xbd8430bd08277231, 0x50c6ff782a838353,
+      0xece53cec4a314ebd, 0xa4f8bf5635246428,
+      0x940f4613ae5ed136, 0x871b7795e136be99,
+      0xb913179899f68584, 0x28e2557b59846e3f,
+      0xe757dd7ec07426e5, 0x331aeada2fe589cf,
+      0x9096ea6f3848984f, 0x3ff0d2c85def7621,
+      0xb4bca50b065abe63, 0xfed077a756b53a9,
+      0xe1ebce4dc7f16dfb, 0xd3e8495912c62894,
+      0x8d3360f09cf6e4bd, 0x64712dd7abbbd95c,
+      0xb080392cc4349dec, 0xbd8d794d96aacfb3,
+      0xdca04777f541c567, 0xecf0d7a0fc5583a0,
+      0x89e42caaf9491b60, 0xf41686c49db57244,
+      0xac5d37d5b79b6239, 0x311c2875c522ced5,
+      0xd77485cb25823ac7, 0x7d633293366b828b,
+      0x86a8d39ef77164bc, 0xae5dff9c02033197,
+      0xa8530886b54dbdeb, 0xd9f57f830283fdfc,
+      0xd267caa862a12d66, 0xd072df63c324fd7b,
+      0x8380dea93da4bc60, 0x4247cb9e59f71e6d,
+      0xa46116538d0deb78, 0x52d9be85f074e608,
+      0xcd795be870516656, 0x67902e276c921f8b,
+      0x806bd9714632dff6, 0xba1cd8a3db53b6,
+      0xa086cfcd97bf97f3, 0x80e8a40eccd228a4,
+      0xc8a883c0fdaf7df0, 0x6122cd128006b2cd,
+      0xfad2a4b13d1b5d6c, 0x796b805720085f81,
+      0x9cc3a6eec6311a63, 0xcbe3303674053bb0,
+      0xc3f490aa77bd60fc, 0xbedbfc4411068a9c,
+      0xf4f1b4d515acb93b, 0xee92fb5515482d44,
+      0x991711052d8bf3c5, 0x751bdd152d4d1c4a,
+      0xbf5cd54678eef0b6, 0xd262d45a78a0635d,
+      0xef340a98172aace4, 0x86fb897116c87c34,
+      0x9580869f0e7aac0e, 0xd45d35e6ae3d4da0,
+      0xbae0a846d2195712, 0x8974836059cca109,
+      0xe998d258869facd7, 0x2bd1a438703fc94b,
+      0x91ff83775423cc06, 0x7b6306a34627ddcf,
+      0xb67f6455292cbf08, 0x1a3bc84c17b1d542,
+      0xe41f3d6a7377eeca, 0x20caba5f1d9e4a93,
+      0x8e938662882af53e, 0x547eb47b7282ee9c,
+      0xb23867fb2a35b28d, 0xe99e619a4f23aa43,
+      0xdec681f9f4c31f31, 0x6405fa00e2ec94d4,
+      0x8b3c113c38f9f37e, 0xde83bc408dd3dd04,
+      0xae0b158b4738705e, 0x9624ab50b148d445,
+      0xd98ddaee19068c76, 0x3badd624dd9b0957,
+      0x87f8a8d4cfa417c9, 0xe54ca5d70a80e5d6,
+      0xa9f6d30a038d1dbc, 0x5e9fcf4ccd211f4c,
+      0xd47487cc8470652b, 0x7647c3200069671f,
+      0x84c8d4dfd2c63f3b, 0x29ecd9f40041e073,
+      0xa5fb0a17c777cf09, 0xf468107100525890,
+      0xcf79cc9db955c2cc, 0x7182148d4066eeb4,
+      0x81ac1fe293d599bf, 0xc6f14cd848405530,
+      0xa21727db38cb002f, 0xb8ada00e5a506a7c,
+      0xca9cf1d206fdc03b, 0xa6d90811f0e4851c,
+      0xfd442e4688bd304a, 0x908f4a166d1da663,
+      0x9e4a9cec15763e2e, 0x9a598e4e043287fe,
+      0xc5dd44271ad3cdba, 0x40eff1e1853f29fd,
+      0xf7549530e188c128, 0xd12bee59e68ef47c,
+      0x9a94dd3e8cf578b9, 0x82bb74f8301958ce,
+      0xc13a148e3032d6e7, 0xe36a52363c1faf01,
+      0xf18899b1bc3f8ca1, 0xdc44e6c3cb279ac1,
+      0x96f5600f15a7b7e5, 0x29ab103a5ef8c0b9,
+      0xbcb2b812db11a5de, 0x7415d448f6b6f0e7,
+      0xebdf661791d60f56, 0x111b495b3464ad21,
+      0x936b9fcebb25c995, 0xcab10dd900beec34,
+      0xb84687c269ef3bfb, 0x3d5d514f40eea742,
+      0xe65829b3046b0afa, 0xcb4a5a3112a5112,
+      0x8ff71a0fe2c2e6dc, 0x47f0e785eaba72ab,
+      0xb3f4e093db73a093, 0x59ed216765690f56,
+      0xe0f218b8d25088b8, 0x306869c13ec3532c,
+      0x8c974f7383725573, 0x1e414218c73a13fb,
+      0xafbd2350644eeacf, 0xe5d1929ef90898fa,
+      0xdbac6c247d62a583, 0xdf45f746b74abf39,
+      0x894bc396ce5da772, 0x6b8bba8c328eb783,
+      0xab9eb47c81f5114f, 0x66ea92f3f326564,
+      0xd686619ba27255a2, 0xc80a537b0efefebd,
+      0x8613fd0145877585, 0xbd06742ce95f5f36,
+      0xa798fc4196e952e7, 0x2c48113823b73704,
+      0xd17f3b51fca3a7a0, 0xf75a15862ca504c5,
+      0x82ef85133de648c4, 0x9a984d73dbe722fb,
+      0xa3ab66580d5fdaf5, 0xc13e60d0d2e0ebba,
+      0xcc963fee10b7d1b3, 0x318df905079926a8,
+      0xffbbcfe994e5c61f, 0xfdf17746497f7052,
+      0x9fd561f1fd0f9bd3, 0xfeb6ea8bedefa633,
+      0xc7caba6e7c5382c8, 0xfe64a52ee96b8fc0,
+      0xf9bd690a1b68637b, 0x3dfdce7aa3c673b0,
+      0x9c1661a651213e2d, 0x6bea10ca65c084e,
+      0xc31bfa0fe5698db8, 0x486e494fcff30a62,
+      0xf3e2f893dec3f126, 0x5a89dba3c3efccfa,
+      0x986ddb5c6b3a76b7, 0xf89629465a75e01c,
+      0xbe89523386091465, 0xf6bbb397f1135823,
+      0xee2ba6c0678b597f, 0x746aa07ded582e2c,
+      0x94db483840b717ef, 0xa8c2a44eb4571cdc,
+      0xba121a4650e4ddeb, 0x92f34d62616ce413,
+      0xe896a0d7e51e1566, 0x77b020baf9c81d17,
+      0x915e2486ef32cd60, 0xace1474dc1d122e,
+      0xb5b5ada8aaff80b8, 0xd819992132456ba,
+      0xe3231912d5bf60e6, 0x10e1fff697ed6c69,
+      0x8df5efabc5979c8f, 0xca8d3ffa1ef463c1,
+      0xb1736b96b6fd83b3, 0xbd308ff8a6b17cb2,
+      0xddd0467c64bce4a0, 0xac7cb3f6d05ddbde,
+      0x8aa22c0dbef60ee4, 0x6bcdf07a423aa96b,
+      0xad4ab7112eb3929d, 0x86c16c98d2c953c6,
+      0xd89d64d57a607744, 0xe871c7bf077ba8b7,
+      0x87625f056c7c4a8b, 0x11471cd764ad4972,
+      0xa93af6c6c79b5d2d, 0xd598e40d3dd89bcf,
+      0xd389b47879823479, 0x4aff1d108d4ec2c3,
+      0x843610cb4bf160cb, 0xcedf722a585139ba,
+      0xa54394fe1eedb8fe, 0xc2974eb4ee658828,
+      0xce947a3da6a9273e, 0x733d226229feea32,
+      0x811ccc668829b887, 0x806357d5a3f525f,
+      0xa163ff802a3426a8, 0xca07c2dcb0cf26f7,
+      0xc9bcff6034c13052, 0xfc89b393dd02f0b5,
+      0xfc2c3f3841f17c67, 0xbbac2078d443ace2,
+      0x9d9ba7832936edc0, 0xd54b944b84aa4c0d,
+      0xc5029163f384a931, 0xa9e795e65d4df11,
+      0xf64335bcf065d37d, 0x4d4617b5ff4a16d5,
+      0x99ea0196163fa42e, 0x504bced1bf8e4e45,
+      0xc06481fb9bcf8d39, 0xe45ec2862f71e1d6,
+      0xf07da27a82c37088, 0x5d767327bb4e5a4c,
+      0x964e858c91ba2655, 0x3a6a07f8d510f86f,
+      0xbbe226efb628afea, 0x890489f70a55368b,
+      0xeadab0aba3b2dbe5, 0x2b45ac74ccea842e,
+      0x92c8ae6b464fc96f, 0x3b0b8bc90012929d,
+      0xb77ada0617e3bbcb, 0x9ce6ebb40173744,
+      0xe55990879ddcaabd, 0xcc420a6a101d0515,
+      0x8f57fa54c2a9eab6, 0x9fa946824a12232d,
+      0xb32df8e9f3546564, 0x47939822dc96abf9,
+      0xdff9772470297ebd, 0x59787e2b93bc56f7,
+      0x8bfbea76c619ef36, 0x57eb4edb3c55b65a,
+      0xaefae51477a06b03, 0xede622920b6b23f1,
+      0xdab99e59958885c4, 0xe95fab368e45eced,
+      0x88b402f7fd75539b, 0x11dbcb0218ebb414,
+      0xaae103b5fcd2a881, 0xd652bdc29f26a119,
+      0xd59944a37c0752a2, 0x4be76d3346f0495f,
+      0x857fcae62d8493a5, 0x6f70a4400c562ddb,
+      0xa6dfbd9fb8e5b88e, 0xcb4ccd500f6bb952,
+      0xd097ad07a71f26b2, 0x7e2000a41346a7a7,
+      0x825ecc24c873782f, 0x8ed400668c0c28c8,
+      0xa2f67f2dfa90563b, 0x728900802f0f32fa,
+      0xcbb41ef979346bca, 0x4f2b40a03ad2ffb9,
+      0xfea126b7d78186bc, 0xe2f610c84987bfa8,
+      0x9f24b832e6b0f436, 0xdd9ca7d2df4d7c9,
+      0xc6ede63fa05d3143, 0x91503d1c79720dbb,
+      0xf8a95fcf88747d94, 0x75a44c6397ce912a,
+      0x9b69dbe1b548ce7c, 0xc986afbe3ee11aba,
+      0xc24452da229b021b, 0xfbe85badce996168,
+      0xf2d56790ab41c2a2, 0xfae27299423fb9c3,
+      0x97c560ba6b0919a5, 0xdccd879fc967d41a,
+      0xbdb6b8e905cb600f, 0x5400e987bbc1c920,
+      0xed246723473e3813, 0x290123e9aab23b68,
+      0x9436c0760c86e30b, 0xf9a0b6720aaf6521,
+      0xb94470938fa89bce, 0xf808e40e8d5b3e69,
+      0xe7958cb87392c2c2, 0xb60b1d1230b20e04,
+      0x90bd77f3483bb9b9, 0xb1c6f22b5e6f48c2,
+      0xb4ecd5f01a4aa828, 0x1e38aeb6360b1af3,
+      0xe2280b6c20dd5232, 0x25c6da63c38de1b0,
+      0x8d590723948a535f, 0x579c487e5a38ad0e,
+      0xb0af48ec79ace837, 0x2d835a9df0c6d851,
+      0xdcdb1b2798182244, 0xf8e431456cf88e65,
+      0x8a08f0f8bf0f156b, 0x1b8e9ecb641b58ff,
+      0xac8b2d36eed2dac5, 0xe272467e3d222f3f,
+      0xd7adf884aa879177, 0x5b0ed81dcc6abb0f,
+      0x86ccbb52ea94baea, 0x98e947129fc2b4e9,
+      0xa87fea27a539e9a5, 0x3f2398d747b36224,
+      0xd29fe4b18e88640e, 0x8eec7f0d19a03aad,
+      0x83a3eeeef9153e89, 0x1953cf68300424ac,
+      0xa48ceaaab75a8e2b, 0x5fa8c3423c052dd7,
+      0xcdb02555653131b6, 0x3792f412cb06794d,
+      0x808e17555f3ebf11, 0xe2bbd88bbee40bd0,
+      0xa0b19d2ab70e6ed6, 0x5b6aceaeae9d0ec4,
+      0xc8de047564d20a8b, 0xf245825a5a445275,
+      0xfb158592be068d2e, 0xeed6e2f0f0d56712,
+      0x9ced737bb6c4183d, 0x55464dd69685606b,
+      0xc428d05aa4751e4c, 0xaa97e14c3c26b886,
+      0xf53304714d9265df, 0xd53dd99f4b3066a8,
+      0x993fe2c6d07b7fab, 0xe546a8038efe4029,
+      0xbf8fdb78849a5f96, 0xde98520472bdd033,
+      0xef73d256a5c0f77c, 0x963e66858f6d4440,
+      0x95a8637627989aad, 0xdde7001379a44aa8,
+      0xbb127c53b17ec159, 0x5560c018580d5d52,
+      0xe9d71b689dde71af, 0xaab8f01e6e10b4a6,
+      0x9226712162ab070d, 0xcab3961304ca70e8,
+      0xb6b00d69bb55c8d1, 0x3d607b97c5fd0d22,
+      0xe45c10c42a2b3b05, 0x8cb89a7db77c506a,
+      0x8eb98a7a9a5b04e3, 0x77f3608e92adb242,
+      0xb267ed1940f1c61c, 0x55f038b237591ed3,
+      0xdf01e85f912e37a3, 0x6b6c46dec52f6688,
+      0x8b61313bbabce2c6, 0x2323ac4b3b3da015,
+      0xae397d8aa96c1b77, 0xabec975e0a0d081a,
+      0xd9c7dced53c72255, 0x96e7bd358c904a21,
+      0x881cea14545c7575, 0x7e50d64177da2e54,
+      0xaa242499697392d2, 0xdde50bd1d5d0b9e9,
+      0xd4ad2dbfc3d07787, 0x955e4ec64b44e864,
+      0x84ec3c97da624ab4, 0xbd5af13bef0b113e,
+      0xa6274bbdd0fadd61, 0xecb1ad8aeacdd58e,
+      0xcfb11ead453994ba, 0x67de18eda5814af2,
+      0x81ceb32c4b43fcf4, 0x80eacf948770ced7,
+      0xa2425ff75e14fc31, 0xa1258379a94d028d,
+      0xcad2f7f5359a3b3e, 0x96ee45813a04330,
+      0xfd87b5f28300ca0d, 0x8bca9d6e188853fc,
+      0x9e74d1b791e07e48, 0x775ea264cf55347e,
+      0xc612062576589dda, 0x95364afe032a819e,
+      0xf79687aed3eec551, 0x3a83ddbd83f52205,
+      0x9abe14cd44753b52, 0xc4926a9672793543,
+      0xc16d9a0095928a27, 0x75b7053c0f178294,
+      0xf1c90080baf72cb1, 0x5324c68b12dd6339,
+      0x971da05074da7bee, 0xd3f6fc16ebca5e04,
+      0xbce5086492111aea, 0x88f4bb1ca6bcf585,
+      0xec1e4a7db69561a5, 0x2b31e9e3d06c32e6,
+      0x9392ee8e921d5d07, 0x3aff322e62439fd0,
+      0xb877aa3236a4b449, 0x9befeb9fad487c3,
+      0xe69594bec44de15b, 0x4c2ebe687989a9b4,
+      0x901d7cf73ab0acd9, 0xf9d37014bf60a11,
+      0xb424dc35095cd80f, 0x538484c19ef38c95,
+      0xe12e13424bb40e13, 0x2865a5f206b06fba,
+      0x8cbccc096f5088cb, 0xf93f87b7442e45d4,
+      0xafebff0bcb24aafe, 0xf78f69a51539d749,
+      0xdbe6fecebdedd5be, 0xb573440e5a884d1c,
+      0x89705f4136b4a597, 0x31680a88f8953031,
+      0xabcc77118461cefc, 0xfdc20d2b36ba7c3e,
+      0xd6bf94d5e57a42bc, 0x3d32907604691b4d,
+      0x8637bd05af6c69b5, 0xa63f9a49c2c1b110,
+      0xa7c5ac471b478423, 0xfcf80dc33721d54,
+      0xd1b71758e219652b, 0xd3c36113404ea4a9,
+      0x83126e978d4fdf3b, 0x645a1cac083126ea,
+      0xa3d70a3d70a3d70a, 0x3d70a3d70a3d70a4,
+      0xcccccccccccccccc, 0xcccccccccccccccd,
+      0x8000000000000000, 0x0,
+      0xa000000000000000, 0x0,
+      0xc800000000000000, 0x0,
+      0xfa00000000000000, 0x0,
+      0x9c40000000000000, 0x0,
+      0xc350000000000000, 0x0,
+      0xf424000000000000, 0x0,
+      0x9896800000000000, 0x0,
+      0xbebc200000000000, 0x0,
+      0xee6b280000000000, 0x0,
+      0x9502f90000000000, 0x0,
+      0xba43b74000000000, 0x0,
+      0xe8d4a51000000000, 0x0,
+      0x9184e72a00000000, 0x0,
+      0xb5e620f480000000, 0x0,
+      0xe35fa931a0000000, 0x0,
+      0x8e1bc9bf04000000, 0x0,
+      0xb1a2bc2ec5000000, 0x0,
+      0xde0b6b3a76400000, 0x0,
+      0x8ac7230489e80000, 0x0,
+      0xad78ebc5ac620000, 0x0,
+      0xd8d726b7177a8000, 0x0,
+      0x878678326eac9000, 0x0,
+      0xa968163f0a57b400, 0x0,
+      0xd3c21bcecceda100, 0x0,
+      0x84595161401484a0, 0x0,
+      0xa56fa5b99019a5c8, 0x0,
+      0xcecb8f27f4200f3a, 0x0,
+      0x813f3978f8940984, 0x4000000000000000,
+      0xa18f07d736b90be5, 0x5000000000000000,
+      0xc9f2c9cd04674ede, 0xa400000000000000,
+      0xfc6f7c4045812296, 0x4d00000000000000,
+      0x9dc5ada82b70b59d, 0xf020000000000000,
+      0xc5371912364ce305, 0x6c28000000000000,
+      0xf684df56c3e01bc6, 0xc732000000000000,
+      0x9a130b963a6c115c, 0x3c7f400000000000,
+      0xc097ce7bc90715b3, 0x4b9f100000000000,
+      0xf0bdc21abb48db20, 0x1e86d40000000000,
+      0x96769950b50d88f4, 0x1314448000000000,
+      0xbc143fa4e250eb31, 0x17d955a000000000,
+      0xeb194f8e1ae525fd, 0x5dcfab0800000000,
+      0x92efd1b8d0cf37be, 0x5aa1cae500000000,
+      0xb7abc627050305ad, 0xf14a3d9e40000000,
+      0xe596b7b0c643c719, 0x6d9ccd05d0000000,
+      0x8f7e32ce7bea5c6f, 0xe4820023a2000000,
+      0xb35dbf821ae4f38b, 0xdda2802c8a800000,
+      0xe0352f62a19e306e, 0xd50b2037ad200000,
+      0x8c213d9da502de45, 0x4526f422cc340000,
+      0xaf298d050e4395d6, 0x9670b12b7f410000,
+      0xdaf3f04651d47b4c, 0x3c0cdd765f114000,
+      0x88d8762bf324cd0f, 0xa5880a69fb6ac800,
+      0xab0e93b6efee0053, 0x8eea0d047a457a00,
+      0xd5d238a4abe98068, 0x72a4904598d6d880,
+      0x85a36366eb71f041, 0x47a6da2b7f864750,
+      0xa70c3c40a64e6c51, 0x999090b65f67d924,
+      0xd0cf4b50cfe20765, 0xfff4b4e3f741cf6d,
+      0x82818f1281ed449f, 0xbff8f10e7a8921a4,
+      0xa321f2d7226895c7, 0xaff72d52192b6a0d,
+      0xcbea6f8ceb02bb39, 0x9bf4f8a69f764490,
+      0xfee50b7025c36a08, 0x2f236d04753d5b4,
+      0x9f4f2726179a2245, 0x1d762422c946590,
+      0xc722f0ef9d80aad6, 0x424d3ad2b7b97ef5,
+      0xf8ebad2b84e0d58b, 0xd2e0898765a7deb2,
+      0x9b934c3b330c8577, 0x63cc55f49f88eb2f,
+      0xc2781f49ffcfa6d5, 0x3cbf6b71c76b25fb,
+      0xf316271c7fc3908a, 0x8bef464e3945ef7a,
+      0x97edd871cfda3a56, 0x97758bf0e3cbb5ac,
+      0xbde94e8e43d0c8ec, 0x3d52eeed1cbea317,
+      0xed63a231d4c4fb27, 0x4ca7aaa863ee4bdd,
+      0x945e455f24fb1cf8, 0x8fe8caa93e74ef6a,
+      0xb975d6b6ee39e436, 0xb3e2fd538e122b44,
+      0xe7d34c64a9c85d44, 0x60dbbca87196b616,
+      0x90e40fbeea1d3a4a, 0xbc8955e946fe31cd,
+      0xb51d13aea4a488dd, 0x6babab6398bdbe41,
+      0xe264589a4dcdab14, 0xc696963c7eed2dd1,
+      0x8d7eb76070a08aec, 0xfc1e1de5cf543ca2,
+      0xb0de65388cc8ada8, 0x3b25a55f43294bcb,
+      0xdd15fe86affad912, 0x49ef0eb713f39ebe,
+      0x8a2dbf142dfcc7ab, 0x6e3569326c784337,
+      0xacb92ed9397bf996, 0x49c2c37f07965404,
+      0xd7e77a8f87daf7fb, 0xdc33745ec97be906,
+      0x86f0ac99b4e8dafd, 0x69a028bb3ded71a3,
+      0xa8acd7c0222311bc, 0xc40832ea0d68ce0c,
+      0xd2d80db02aabd62b, 0xf50a3fa490c30190,
+      0x83c7088e1aab65db, 0x792667c6da79e0fa,
+      0xa4b8cab1a1563f52, 0x577001b891185938,
+      0xcde6fd5e09abcf26, 0xed4c0226b55e6f86,
+      0x80b05e5ac60b6178, 0x544f8158315b05b4,
+      0xa0dc75f1778e39d6, 0x696361ae3db1c721,
+      0xc913936dd571c84c, 0x3bc3a19cd1e38e9,
+      0xfb5878494ace3a5f, 0x4ab48a04065c723,
+      0x9d174b2dcec0e47b, 0x62eb0d64283f9c76,
+      0xc45d1df942711d9a, 0x3ba5d0bd324f8394,
+      0xf5746577930d6500, 0xca8f44ec7ee36479,
+      0x9968bf6abbe85f20, 0x7e998b13cf4e1ecb,
+      0xbfc2ef456ae276e8, 0x9e3fedd8c321a67e,
+      0xefb3ab16c59b14a2, 0xc5cfe94ef3ea101e,
+      0x95d04aee3b80ece5, 0xbba1f1d158724a12,
+      0xbb445da9ca61281f, 0x2a8a6e45ae8edc97,
+      0xea1575143cf97226, 0xf52d09d71a3293bd,
+      0x924d692ca61be758, 0x593c2626705f9c56,
+      0xb6e0c377cfa2e12e, 0x6f8b2fb00c77836c,
+      0xe498f455c38b997a, 0xb6dfb9c0f956447,
+      0x8edf98b59a373fec, 0x4724bd4189bd5eac,
+      0xb2977ee300c50fe7, 0x58edec91ec2cb657,
+      0xdf3d5e9bc0f653e1, 0x2f2967b66737e3ed,
+      0x8b865b215899f46c, 0xbd79e0d20082ee74,
+      0xae67f1e9aec07187, 0xecd8590680a3aa11,
+      0xda01ee641a708de9, 0xe80e6f4820cc9495,
+      0x884134fe908658b2, 0x3109058d147fdcdd,
+      0xaa51823e34a7eede, 0xbd4b46f0599fd415,
+      0xd4e5e2cdc1d1ea96, 0x6c9e18ac7007c91a,
+      0x850fadc09923329e, 0x3e2cf6bc604ddb0,
+      0xa6539930bf6bff45, 0x84db8346b786151c,
+      0xcfe87f7cef46ff16, 0xe612641865679a63,
+      0x81f14fae158c5f6e, 0x4fcb7e8f3f60c07e,
+      0xa26da3999aef7749, 0xe3be5e330f38f09d,
+      0xcb090c8001ab551c, 0x5cadf5bfd3072cc5,
+      0xfdcb4fa002162a63, 0x73d9732fc7c8f7f6,
+      0x9e9f11c4014dda7e, 0x2867e7fddcdd9afa,
+      0xc646d63501a1511d, 0xb281e1fd541501b8,
+      0xf7d88bc24209a565, 0x1f225a7ca91a4226,
+      0x9ae757596946075f, 0x3375788de9b06958,
+      0xc1a12d2fc3978937, 0x52d6b1641c83ae,
+      0xf209787bb47d6b84, 0xc0678c5dbd23a49a,
+      0x9745eb4d50ce6332, 0xf840b7ba963646e0,
+      0xbd176620a501fbff, 0xb650e5a93bc3d898,
+      0xec5d3fa8ce427aff, 0xa3e51f138ab4cebe,
+      0x93ba47c980e98cdf, 0xc66f336c36b10137,
+      0xb8a8d9bbe123f017, 0xb80b0047445d4184,
+      0xe6d3102ad96cec1d, 0xa60dc059157491e5,
+      0x9043ea1ac7e41392, 0x87c89837ad68db2f,
+      0xb454e4a179dd1877, 0x29babe4598c311fb,
+      0xe16a1dc9d8545e94, 0xf4296dd6fef3d67a,
+      0x8ce2529e2734bb1d, 0x1899e4a65f58660c,
+      0xb01ae745b101e9e4, 0x5ec05dcff72e7f8f,
+      0xdc21a1171d42645d, 0x76707543f4fa1f73,
+      0x899504ae72497eba, 0x6a06494a791c53a8,
+      0xabfa45da0edbde69, 0x487db9d17636892,
+      0xd6f8d7509292d603, 0x45a9d2845d3c42b6,
+      0x865b86925b9bc5c2, 0xb8a2392ba45a9b2,
+      0xa7f26836f282b732, 0x8e6cac7768d7141e,
+      0xd1ef0244af2364ff, 0x3207d795430cd926,
+      0x8335616aed761f1f, 0x7f44e6bd49e807b8,
+      0xa402b9c5a8d3a6e7, 0x5f16206c9c6209a6,
+      0xcd036837130890a1, 0x36dba887c37a8c0f,
+      0x802221226be55a64, 0xc2494954da2c9789,
+      0xa02aa96b06deb0fd, 0xf2db9baa10b7bd6c,
+      0xc83553c5c8965d3d, 0x6f92829494e5acc7,
+      0xfa42a8b73abbf48c, 0xcb772339ba1f17f9,
+      0x9c69a97284b578d7, 0xff2a760414536efb,
+      0xc38413cf25e2d70d, 0xfef5138519684aba,
+      0xf46518c2ef5b8cd1, 0x7eb258665fc25d69,
+      0x98bf2f79d5993802, 0xef2f773ffbd97a61,
+      0xbeeefb584aff8603, 0xaafb550ffacfd8fa,
+      0xeeaaba2e5dbf6784, 0x95ba2a53f983cf38,
+      0x952ab45cfa97a0b2, 0xdd945a747bf26183,
+      0xba756174393d88df, 0x94f971119aeef9e4,
+      0xe912b9d1478ceb17, 0x7a37cd5601aab85d,
+      0x91abb422ccb812ee, 0xac62e055c10ab33a,
+      0xb616a12b7fe617aa, 0x577b986b314d6009,
+      0xe39c49765fdf9d94, 0xed5a7e85fda0b80b,
+      0x8e41ade9fbebc27d, 0x14588f13be847307,
+      0xb1d219647ae6b31c, 0x596eb2d8ae258fc8,
+      0xde469fbd99a05fe3, 0x6fca5f8ed9aef3bb,
+      0x8aec23d680043bee, 0x25de7bb9480d5854,
+      0xada72ccc20054ae9, 0xaf561aa79a10ae6a,
+      0xd910f7ff28069da4, 0x1b2ba1518094da04,
+      0x87aa9aff79042286, 0x90fb44d2f05d0842,
+      0xa99541bf57452b28, 0x353a1607ac744a53,
+      0xd3fa922f2d1675f2, 0x42889b8997915ce8,
+      0x847c9b5d7c2e09b7, 0x69956135febada11,
+      0xa59bc234db398c25, 0x43fab9837e699095,
+      0xcf02b2c21207ef2e, 0x94f967e45e03f4bb,
+      0x8161afb94b44f57d, 0x1d1be0eebac278f5,
+      0xa1ba1ba79e1632dc, 0x6462d92a69731732,
+      0xca28a291859bbf93, 0x7d7b8f7503cfdcfe,
+      0xfcb2cb35e702af78, 0x5cda735244c3d43e,
+      0x9defbf01b061adab, 0x3a0888136afa64a7,
+      0xc56baec21c7a1916, 0x88aaa1845b8fdd0,
+      0xf6c69a72a3989f5b, 0x8aad549e57273d45,
+      0x9a3c2087a63f6399, 0x36ac54e2f678864b,
+      0xc0cb28a98fcf3c7f, 0x84576a1bb416a7dd,
+      0xf0fdf2d3f3c30b9f, 0x656d44a2a11c51d5,
+      0x969eb7c47859e743, 0x9f644ae5a4b1b325,
+      0xbc4665b596706114, 0x873d5d9f0dde1fee,
+      0xeb57ff22fc0c7959, 0xa90cb506d155a7ea,
+      0x9316ff75dd87cbd8, 0x9a7f12442d588f2,
+      0xb7dcbf5354e9bece, 0xc11ed6d538aeb2f,
+      0xe5d3ef282a242e81, 0x8f1668c8a86da5fa,
+      0x8fa475791a569d10, 0xf96e017d694487bc,
+      0xb38d92d760ec4455, 0x37c981dcc395a9ac,
+      0xe070f78d3927556a, 0x85bbe253f47b1417,
+      0x8c469ab843b89562, 0x93956d7478ccec8e,
+      0xaf58416654a6babb, 0x387ac8d1970027b2,
+      0xdb2e51bfe9d0696a, 0x6997b05fcc0319e,
+      0x88fcf317f22241e2, 0x441fece3bdf81f03,
+      0xab3c2fddeeaad25a, 0xd527e81cad7626c3,
+      0xd60b3bd56a5586f1, 0x8a71e223d8d3b074,
+      0x85c7056562757456, 0xf6872d5667844e49,
+      0xa738c6bebb12d16c, 0xb428f8ac016561db,
+      0xd106f86e69d785c7, 0xe13336d701beba52,
+      0x82a45b450226b39c, 0xecc0024661173473,
+      0xa34d721642b06084, 0x27f002d7f95d0190,
+      0xcc20ce9bd35c78a5, 0x31ec038df7b441f4,
+      0xff290242c83396ce, 0x7e67047175a15271,
+      0x9f79a169bd203e41, 0xf0062c6e984d386,
+      0xc75809c42c684dd1, 0x52c07b78a3e60868,
+      0xf92e0c3537826145, 0xa7709a56ccdf8a82,
+      0x9bbcc7a142b17ccb, 0x88a66076400bb691,
+      0xc2abf989935ddbfe, 0x6acff893d00ea435,
+      0xf356f7ebf83552fe, 0x583f6b8c4124d43,
+      0x98165af37b2153de, 0xc3727a337a8b704a,
+      0xbe1bf1b059e9a8d6, 0x744f18c0592e4c5c,
+      0xeda2ee1c7064130c, 0x1162def06f79df73,
+      0x9485d4d1c63e8be7, 0x8addcb5645ac2ba8,
+      0xb9a74a0637ce2ee1, 0x6d953e2bd7173692,
+      0xe8111c87c5c1ba99, 0xc8fa8db6ccdd0437,
+      0x910ab1d4db9914a0, 0x1d9c9892400a22a2,
+      0xb54d5e4a127f59c8, 0x2503beb6d00cab4b,
+      0xe2a0b5dc971f303a, 0x2e44ae64840fd61d,
+      0x8da471a9de737e24, 0x5ceaecfed289e5d2,
+      0xb10d8e1456105dad, 0x7425a83e872c5f47,
+      0xdd50f1996b947518, 0xd12f124e28f77719,
+      0x8a5296ffe33cc92f, 0x82bd6b70d99aaa6f,
+      0xace73cbfdc0bfb7b, 0x636cc64d1001550b,
+      0xd8210befd30efa5a, 0x3c47f7e05401aa4e,
+      0x8714a775e3e95c78, 0x65acfaec34810a71,
+      0xa8d9d1535ce3b396, 0x7f1839a741a14d0d,
+      0xd31045a8341ca07c, 0x1ede48111209a050,
+      0x83ea2b892091e44d, 0x934aed0aab460432,
+      0xa4e4b66b68b65d60, 0xf81da84d5617853f,
+      0xce1de40642e3f4b9, 0x36251260ab9d668e,
+      0x80d2ae83e9ce78f3, 0xc1d72b7c6b426019,
+      0xa1075a24e4421730, 0xb24cf65b8612f81f,
+      0xc94930ae1d529cfc, 0xdee033f26797b627,
+      0xfb9b7cd9a4a7443c, 0x169840ef017da3b1,
+      0x9d412e0806e88aa5, 0x8e1f289560ee864e,
+      0xc491798a08a2ad4e, 0xf1a6f2bab92a27e2,
+      0xf5b5d7ec8acb58a2, 0xae10af696774b1db,
+      0x9991a6f3d6bf1765, 0xacca6da1e0a8ef29,
+      0xbff610b0cc6edd3f, 0x17fd090a58d32af3,
+      0xeff394dcff8a948e, 0xddfc4b4cef07f5b0,
+      0x95f83d0a1fb69cd9, 0x4abdaf101564f98e,
+      0xbb764c4ca7a4440f, 0x9d6d1ad41abe37f1,
+      0xea53df5fd18d5513, 0x84c86189216dc5ed,
+      0x92746b9be2f8552c, 0x32fd3cf5b4e49bb4,
+      0xb7118682dbb66a77, 0x3fbc8c33221dc2a1,
+      0xe4d5e82392a40515, 0xfabaf3feaa5334a,
+      0x8f05b1163ba6832d, 0x29cb4d87f2a7400e,
+      0xb2c71d5bca9023f8, 0x743e20e9ef511012,
+      0xdf78e4b2bd342cf6, 0x914da9246b255416,
+      0x8bab8eefb6409c1a, 0x1ad089b6c2f7548e,
+      0xae9672aba3d0c320, 0xa184ac2473b529b1,
+      0xda3c0f568cc4f3e8, 0xc9e5d72d90a2741e,
+      0x8865899617fb1871, 0x7e2fa67c7a658892,
+      0xaa7eebfb9df9de8d, 0xddbb901b98feeab7,
+      0xd51ea6fa85785631, 0x552a74227f3ea565,
+      0x8533285c936b35de, 0xd53a88958f87275f,
+      0xa67ff273b8460356, 0x8a892abaf368f137,
+      0xd01fef10a657842c, 0x2d2b7569b0432d85,
+      0x8213f56a67f6b29b, 0x9c3b29620e29fc73,
+      0xa298f2c501f45f42, 0x8349f3ba91b47b8f,
+      0xcb3f2f7642717713, 0x241c70a936219a73,
+      0xfe0efb53d30dd4d7, 0xed238cd383aa0110,
+      0x9ec95d1463e8a506, 0xf4363804324a40aa,
+      0xc67bb4597ce2ce48, 0xb143c6053edcd0d5,
+      0xf81aa16fdc1b81da, 0xdd94b7868e94050a,
+      0x9b10a4e5e9913128, 0xca7cf2b4191c8326,
+      0xc1d4ce1f63f57d72, 0xfd1c2f611f63a3f0,
+      0xf24a01a73cf2dccf, 0xbc633b39673c8cec,
+      0x976e41088617ca01, 0xd5be0503e085d813,
+      0xbd49d14aa79dbc82, 0x4b2d8644d8a74e18,
+      0xec9c459d51852ba2, 0xddf8e7d60ed1219e,
+      0x93e1ab8252f33b45, 0xcabb90e5c942b503,
+      0xb8da1662e7b00a17, 0x3d6a751f3b936243,
+      0xe7109bfba19c0c9d, 0xcc512670a783ad4,
+      0x906a617d450187e2, 0x27fb2b80668b24c5,
+      0xb484f9dc9641e9da, 0xb1f9f660802dedf6,
+      0xe1a63853bbd26451, 0x5e7873f8a0396973,
+      0x8d07e33455637eb2, 0xdb0b487b6423e1e8,
+      0xb049dc016abc5e5f, 0x91ce1a9a3d2cda62,
+      0xdc5c5301c56b75f7, 0x7641a140cc7810fb,
+      0x89b9b3e11b6329ba, 0xa9e904c87fcb0a9d,
+      0xac2820d9623bf429, 0x546345fa9fbdcd44,
+      0xd732290fbacaf133, 0xa97c177947ad4095,
+      0x867f59a9d4bed6c0, 0x49ed8eabcccc485d,
+      0xa81f301449ee8c70, 0x5c68f256bfff5a74,
+      0xd226fc195c6a2f8c, 0x73832eec6fff3111,
+      0x83585d8fd9c25db7, 0xc831fd53c5ff7eab,
+      0xa42e74f3d032f525, 0xba3e7ca8b77f5e55,
+      0xcd3a1230c43fb26f, 0x28ce1bd2e55f35eb,
+      0x80444b5e7aa7cf85, 0x7980d163cf5b81b3,
+      0xa0555e361951c366, 0xd7e105bcc332621f,
+      0xc86ab5c39fa63440, 0x8dd9472bf3fefaa7,
+      0xfa856334878fc150, 0xb14f98f6f0feb951,
+      0x9c935e00d4b9d8d2, 0x6ed1bf9a569f33d3,
+      0xc3b8358109e84f07, 0xa862f80ec4700c8,
+      0xf4a642e14c6262c8, 0xcd27bb612758c0fa,
+      0x98e7e9cccfbd7dbd, 0x8038d51cb897789c,
+      0xbf21e44003acdd2c, 0xe0470a63e6bd56c3,
+      0xeeea5d5004981478, 0x1858ccfce06cac74,
+      0x95527a5202df0ccb, 0xf37801e0c43ebc8,
+      0xbaa718e68396cffd, 0xd30560258f54e6ba,
+      0xe950df20247c83fd, 0x47c6b82ef32a2069,
+      0x91d28b7416cdd27e, 0x4cdc331d57fa5441,
+      0xb6472e511c81471d, 0xe0133fe4adf8e952,
+      0xe3d8f9e563a198e5, 0x58180fddd97723a6,
+      0x8e679c2f5e44ff8f, 0x570f09eaa7ea7648,
+  };
+};
+
+#if SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE
+
+template <class unused>
+constexpr uint64_t
+    powers_template<unused>::power_of_five_128[number_of_entries];
+
+#endif
+
+using powers = powers_template<>;
+
+} // namespace simdjson_fast_float
+
+#endif
+
+#ifndef SIMDJSON_FASTFLOAT_DECIMAL_TO_BINARY_H
+#define SIMDJSON_FASTFLOAT_DECIMAL_TO_BINARY_H
+
+#include <cfloat>
+#include <cinttypes>
+#include <cmath>
+#include <cstdint>
+#include <cstdlib>
+#include <cstring>
+
+namespace simdjson_fast_float {
+
+// This will compute or rather approximate w * 5**q and return a pair of 64-bit
+// words approximating the result, with the "high" part corresponding to the
+// most significant bits and the low part corresponding to the least significant
+// bits.
+//
+template <int bit_precision>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 value128
+compute_product_approximation(int64_t q, uint64_t w) {
+  int const index = 2 * int(q - powers::smallest_power_of_five);
+  // For small values of q, e.g., q in [0,27], the answer is always exact
+  // because The line value128 firstproduct = full_multiplication(w,
+  // power_of_five_128[index]); gives the exact answer.
+  value128 firstproduct =
+      full_multiplication(w, powers::power_of_five_128[index]);
+  static_assert((bit_precision >= 0) && (bit_precision <= 64),
+                " precision should  be in (0,64]");
+  constexpr uint64_t precision_mask =
+      (bit_precision < 64) ? (uint64_t(0xFFFFFFFFFFFFFFFF) >> bit_precision)
+                           : uint64_t(0xFFFFFFFFFFFFFFFF);
+  if ((firstproduct.high & precision_mask) ==
+      precision_mask) { // could further guard with  (lower + w < lower)
+    // regarding the second product, we only need secondproduct.high, but our
+    // expectation is that the compiler will optimize this extra work away if
+    // needed.
+    value128 secondproduct =
+        full_multiplication(w, powers::power_of_five_128[index + 1]);
+    firstproduct.low += secondproduct.high;
+    if (secondproduct.high > firstproduct.low) {
+      firstproduct.high++;
+    }
+  }
+  return firstproduct;
+}
+
+namespace detail {
+/**
+ * For q in (0,350), we have that
+ *  f = (((152170 + 65536) * q ) >> 16);
+ * is equal to
+ *   floor(p) + q
+ * where
+ *   p = log(5**q)/log(2) = q * log(5)/log(2)
+ *
+ * For negative values of q in (-400,0), we have that
+ *  f = (((152170 + 65536) * q ) >> 16);
+ * is equal to
+ *   -ceil(p) + q
+ * where
+ *   p = log(5**-q)/log(2) = -q * log(5)/log(2)
+ */
+constexpr simdjson_fastfloat_really_inline int32_t power(int32_t q) noexcept {
+  return (((152170 + 65536) * q) >> 16) + 63;
+}
+} // namespace detail
+
+// create an adjusted mantissa, biased by the invalid power2
+// for significant digits already multiplied by 10 ** q.
+template <typename binary>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 adjusted_mantissa
+compute_error_scaled(int64_t q, uint64_t w, int lz) noexcept {
+  int hilz = int(w >> 63) ^ 1;
+  adjusted_mantissa answer;
+  answer.mantissa = w << hilz;
+  int bias = binary::mantissa_explicit_bits() - binary::minimum_exponent();
+  answer.power2 = int32_t(detail::power(int32_t(q)) + bias - hilz - lz - 62 +
+                          invalid_am_bias);
+  return answer;
+}
+
+// w * 10 ** q, without rounding the representation up.
+// the power2 in the exponent will be adjusted by invalid_am_bias.
+template <typename binary>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 adjusted_mantissa
+compute_error(int64_t q, uint64_t w) noexcept {
+  int lz = leading_zeroes(w);
+  w <<= lz;
+  value128 product =
+      compute_product_approximation<binary::mantissa_explicit_bits() + 3>(q, w);
+  return compute_error_scaled<binary>(q, product.high, lz);
+}
+
+// Computers w * 10 ** q.
+// The returned value should be a valid number that simply needs to be
+// packed. However, in some very rare cases, the computation will fail. In such
+// cases, we return an adjusted_mantissa with a negative power of 2: the caller
+// should recompute in such cases.
+template <typename binary>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 adjusted_mantissa
+compute_float(int64_t q, uint64_t w) noexcept {
+  adjusted_mantissa answer;
+  if ((w == 0) || (q < binary::smallest_power_of_ten())) {
+    answer.power2 = 0;
+    answer.mantissa = 0;
+    // result should be zero
+    return answer;
+  }
+  if (q > binary::largest_power_of_ten()) {
+    // we want to get infinity:
+    answer.power2 = binary::infinite_power();
+    answer.mantissa = 0;
+    return answer;
+  }
+  // At this point in time q is in [powers::smallest_power_of_five,
+  // powers::largest_power_of_five].
+
+  // We want the most significant bit of i to be 1. Shift if needed.
+  int lz = leading_zeroes(w);
+  w <<= lz;
+
+  // The required precision is binary::mantissa_explicit_bits() + 3 because
+  // 1. We need the implicit bit
+  // 2. We need an extra bit for rounding purposes
+  // 3. We might lose a bit due to the "upperbit" routine (result too small,
+  // requiring a shift)
+
+  value128 product =
+      compute_product_approximation<binary::mantissa_explicit_bits() + 3>(q, w);
+  // The computed 'product' is always sufficient.
+  // Mathematical proof:
+  // Noble Mushtak and Daniel Lemire, Fast Number Parsing Without Fallback (to
+  // appear) See script/mushtak_lemire.py
+
+  // The "compute_product_approximation" function can be slightly slower than a
+  // branchless approach: value128 product = compute_product(q, w); but in
+  // practice, we can win big with the compute_product_approximation if its
+  // additional branch is easily predicted. Which is best is data specific.
+  int upperbit = int(product.high >> 63);
+  int shift = upperbit + 64 - binary::mantissa_explicit_bits() - 3;
+
+  answer.mantissa = product.high >> shift;
+
+  answer.power2 = int32_t(detail::power(int32_t(q)) + upperbit - lz -
+                          binary::minimum_exponent());
+  if (answer.power2 <= 0) { // we have a subnormal?
+    // Here have that answer.power2 <= 0 so -answer.power2 >= 0
+    if (-answer.power2 + 1 >=
+        64) { // if we have more than 64 bits below the minimum exponent, you
+              // have a zero for sure.
+      answer.power2 = 0;
+      answer.mantissa = 0;
+      // result should be zero
+      return answer;
+    }
+    // next line is safe because -answer.power2 + 1 < 64
+    answer.mantissa >>= -answer.power2 + 1;
+    // Thankfully, we can't have both "round-to-even" and subnormals because
+    // "round-to-even" only occurs for powers close to 0 in the 32-bit and
+    // and 64-bit case (with no more than 19 digits).
+    answer.mantissa += (answer.mantissa & 1); // round up
+    answer.mantissa >>= 1;
+    // There is a weird scenario where we don't have a subnormal but just.
+    // Suppose we start with 2.2250738585072013e-308, we end up
+    // with 0x3fffffffffffff x 2^-1023-53 which is technically subnormal
+    // whereas 0x40000000000000 x 2^-1023-53  is normal. Now, we need to round
+    // up 0x3fffffffffffff x 2^-1023-53  and once we do, we are no longer
+    // subnormal, but we can only know this after rounding.
+    // So we only declare a subnormal if we are smaller than the threshold.
+    answer.power2 =
+        (answer.mantissa < (uint64_t(1) << binary::mantissa_explicit_bits()))
+            ? 0
+            : 1;
+    return answer;
+  }
+
+  // usually, we round *up*, but if we fall right in between and and we have an
+  // even basis, we need to round down
+  // We are only concerned with the cases where 5**q fits in single 64-bit word.
+  if ((product.low <= 1) && (q >= binary::min_exponent_round_to_even()) &&
+      (q <= binary::max_exponent_round_to_even()) &&
+      ((answer.mantissa & 3) == 1)) { // we may fall between two floats!
+    // To be in-between two floats we need that in doing
+    //   answer.mantissa = product.high >> (upperbit + 64 -
+    //   binary::mantissa_explicit_bits() - 3);
+    // ... we dropped out only zeroes. But if this happened, then we can go
+    // back!!!
+    if ((answer.mantissa << shift) == product.high) {
+      answer.mantissa &= ~uint64_t(1); // flip it so that we do not round up
+    }
+  }
+
+  answer.mantissa += (answer.mantissa & 1); // round up
+  answer.mantissa >>= 1;
+  if (answer.mantissa >= (uint64_t(2) << binary::mantissa_explicit_bits())) {
+    answer.mantissa = (uint64_t(1) << binary::mantissa_explicit_bits());
+    answer.power2++; // undo previous addition
+  }
+
+  answer.mantissa &= ~(uint64_t(1) << binary::mantissa_explicit_bits());
+  if (answer.power2 >= binary::infinite_power()) { // infinity
+    answer.power2 = binary::infinite_power();
+    answer.mantissa = 0;
+  }
+  return answer;
+}
+
+} // namespace simdjson_fast_float
+
+#endif
+
+#ifndef SIMDJSON_FASTFLOAT_BIGINT_H
+#define SIMDJSON_FASTFLOAT_BIGINT_H
+
+#include <algorithm>
+#include <cstdint>
+#include <climits>
+#include <cstring>
+
+
+namespace simdjson_fast_float {
+
+// the limb width: we want efficient multiplication of double the bits in
+// limb, or for 64-bit limbs, at least 64-bit multiplication where we can
+// extract the high and low parts efficiently. this is every 64-bit
+// architecture except for sparc, which emulates 128-bit multiplication.
+// we might have platforms where `CHAR_BIT` is not 8, so let's avoid
+// doing `8 * sizeof(limb)`.
+#if defined(SIMDJSON_FASTFLOAT_64BIT) && !defined(__sparc)
+#define SIMDJSON_FASTFLOAT_64BIT_LIMB 1
+typedef uint64_t limb;
+constexpr size_t limb_bits = 64;
+#else
+#define SIMDJSON_FASTFLOAT_32BIT_LIMB
+typedef uint32_t limb;
+constexpr size_t limb_bits = 32;
+#endif
+
+typedef span<limb> limb_span;
+
+// number of bits in a bigint. this needs to be at least the number
+// of bits required to store the largest bigint, which is
+// `log2(10**(digits + max_exp))`, or `log2(10**(767 + 342))`, or
+// ~3600 bits, so we round to 4000.
+constexpr size_t bigint_bits = 4000;
+constexpr size_t bigint_limbs = bigint_bits / limb_bits;
+
+// vector-like type that is allocated on the stack. the entire
+// buffer is pre-allocated, and only the length changes.
+template <uint16_t size> struct stackvec {
+  limb data[size];
+  // we never need more than 150 limbs
+  uint16_t length{0};
+
+  stackvec() = default;
+  stackvec(stackvec const &) = delete;
+  stackvec &operator=(stackvec const &) = delete;
+  stackvec(stackvec &&) = delete;
+  stackvec &operator=(stackvec &&other) = delete;
+
+  // create stack vector from existing limb span.
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 stackvec(limb_span s) {
+    SIMDJSON_FASTFLOAT_ASSERT(try_extend(s));
+  }
+
+  SIMDJSON_FASTFLOAT_CONSTEXPR14 limb &operator[](size_t index) noexcept {
+    SIMDJSON_FASTFLOAT_DEBUG_ASSERT(index < length);
+    return data[index];
+  }
+
+  SIMDJSON_FASTFLOAT_CONSTEXPR14 const limb &operator[](size_t index) const noexcept {
+    SIMDJSON_FASTFLOAT_DEBUG_ASSERT(index < length);
+    return data[index];
+  }
+
+  // index from the end of the container
+  SIMDJSON_FASTFLOAT_CONSTEXPR14 const limb &rindex(size_t index) const noexcept {
+    SIMDJSON_FASTFLOAT_DEBUG_ASSERT(index < length);
+    size_t rindex = length - index - 1;
+    return data[rindex];
+  }
+
+  // set the length, without bounds checking.
+  SIMDJSON_FASTFLOAT_CONSTEXPR14 void set_len(size_t len) noexcept {
+    length = uint16_t(len);
+  }
+
+  constexpr size_t len() const noexcept { return length; }
+
+  constexpr bool is_empty() const noexcept { return length == 0; }
+
+  constexpr size_t capacity() const noexcept { return size; }
+
+  // append item to vector, without bounds checking
+  SIMDJSON_FASTFLOAT_CONSTEXPR14 void push_unchecked(limb value) noexcept {
+    data[length] = value;
+    length++;
+  }
+
+  // append item to vector, returning if item was added
+  SIMDJSON_FASTFLOAT_CONSTEXPR14 bool try_push(limb value) noexcept {
+    if (len() < capacity()) {
+      push_unchecked(value);
+      return true;
+    } else {
+      return false;
+    }
+  }
+
+  // add items to the vector, from a span, without bounds checking
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 void extend_unchecked(limb_span s) noexcept {
+    limb *ptr = data + length;
+    std::copy_n(s.ptr, s.len(), ptr);
+    set_len(len() + s.len());
+  }
+
+  // try to add items to the vector, returning if items were added
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 bool try_extend(limb_span s) noexcept {
+    if (len() + s.len() <= capacity()) {
+      extend_unchecked(s);
+      return true;
+    } else {
+      return false;
+    }
+  }
+
+  // resize the vector, without bounds checking
+  // if the new size is longer than the vector, assign value to each
+  // appended item.
+  SIMDJSON_FASTFLOAT_CONSTEXPR20
+  void resize_unchecked(size_t new_len, limb value) noexcept {
+    if (new_len > len()) {
+      size_t count = new_len - len();
+      limb *first = data + len();
+      limb *last = first + count;
+      ::std::fill(first, last, value);
+      set_len(new_len);
+    } else {
+      set_len(new_len);
+    }
+  }
+
+  // try to resize the vector, returning if the vector was resized.
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 bool try_resize(size_t new_len, limb value) noexcept {
+    if (new_len > capacity()) {
+      return false;
+    } else {
+      resize_unchecked(new_len, value);
+      return true;
+    }
+  }
+
+  // check if any limbs are non-zero after the given index.
+  // this needs to be done in reverse order, since the index
+  // is relative to the most significant limbs.
+  SIMDJSON_FASTFLOAT_CONSTEXPR14 bool nonzero(size_t index) const noexcept {
+    while (index < len()) {
+      if (rindex(index) != 0) {
+        return true;
+      }
+      index++;
+    }
+    return false;
+  }
+
+  // normalize the big integer, so most-significant zero limbs are removed.
+  SIMDJSON_FASTFLOAT_CONSTEXPR14 void normalize() noexcept {
+    while (len() > 0 && rindex(0) == 0) {
+      length--;
+    }
+  }
+};
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 uint64_t
+empty_hi64(bool &truncated) noexcept {
+  truncated = false;
+  return 0;
+}

-    std::memmove(buf + (2 + static_cast<size_t>(-n)), buf,
-                 static_cast<size_t>(k));
-    buf[0] = '0';
-    buf[1] = '.';
-    std::memset(buf + 2, '0', static_cast<size_t>(-n));
-    return buf + (2U + static_cast<size_t>(-n) + static_cast<size_t>(k));
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 uint64_t
+uint64_hi64(uint64_t r0, bool &truncated) noexcept {
+  truncated = false;
+  int shl = leading_zeroes(r0);
+  return r0 << shl;
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 uint64_t
+uint64_hi64(uint64_t r0, uint64_t r1, bool &truncated) noexcept {
+  int shl = leading_zeroes(r0);
+  if (shl == 0) {
+    truncated = r1 != 0;
+    return r0;
+  } else {
+    int shr = 64 - shl;
+    truncated = (r1 << shl) != 0;
+    return (r0 << shl) | (r1 >> shr);
   }
+}

-  if (k == 1) {
-    // dE+123
-    // len <= 1 + 5
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 uint64_t
+uint32_hi64(uint32_t r0, bool &truncated) noexcept {
+  return uint64_hi64(r0, truncated);
+}

-    buf += 1;
-  } else {
-    // d.igitsE+123
-    // len <= max_digits10 + 1 + 5
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 uint64_t
+uint32_hi64(uint32_t r0, uint32_t r1, bool &truncated) noexcept {
+  uint64_t x0 = r0;
+  uint64_t x1 = r1;
+  return uint64_hi64((x0 << 32) | x1, truncated);
+}

-    std::memmove(buf + 2, buf + 1, static_cast<size_t>(k) - 1);
-    buf[1] = '.';
-    buf += 1 + static_cast<size_t>(k);
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 uint64_t
+uint32_hi64(uint32_t r0, uint32_t r1, uint32_t r2, bool &truncated) noexcept {
+  uint64_t x0 = r0;
+  uint64_t x1 = r1;
+  uint64_t x2 = r2;
+  return uint64_hi64(x0, (x1 << 32) | x2, truncated);
+}
+
+// add two small integers, checking for overflow.
+// we want an efficient operation. for msvc, where
+// we don't have built-in intrinsics, this is still
+// pretty fast.
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 limb
+scalar_add(limb x, limb y, bool &overflow) noexcept {
+  limb z;
+// gcc and clang
+#if defined(__has_builtin)
+#if __has_builtin(__builtin_add_overflow)
+  if (!cpp20_and_in_constexpr()) {
+    overflow = __builtin_add_overflow(x, y, &z);
+    return z;
   }
+#endif
+#endif

-  *buf++ = 'e';
-  return append_exponent(buf, n - 1);
+  // generic, this still optimizes correctly on MSVC.
+  z = x + y;
+  overflow = z < x;
+  return z;
+}
+
+// multiply two small integers, getting both the high and low bits.
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 limb
+scalar_mul(limb x, limb y, limb &carry) noexcept {
+#ifdef SIMDJSON_FASTFLOAT_64BIT_LIMB
+#if defined(__SIZEOF_INT128__)
+  // GCC and clang both define it as an extension.
+  __uint128_t z = __uint128_t(x) * __uint128_t(y) + __uint128_t(carry);
+  carry = limb(z >> limb_bits);
+  return limb(z);
+#else
+  // fallback, no native 128-bit integer multiplication with carry.
+  // on msvc, this optimizes identically, somehow.
+  value128 z = full_multiplication(x, y);
+  bool overflow;
+  z.low = scalar_add(z.low, carry, overflow);
+  z.high += uint64_t(overflow); // cannot overflow
+  carry = z.high;
+  return z.low;
+#endif
+#else
+  uint64_t z = uint64_t(x) * uint64_t(y) + uint64_t(carry);
+  carry = limb(z >> limb_bits);
+  return limb(z);
+#endif
 }

-} // namespace dtoa_impl
+// add scalar value to bigint starting from offset.
+// used in grade school multiplication
+template <uint16_t size>
+inline SIMDJSON_FASTFLOAT_CONSTEXPR20 bool small_add_from(stackvec<size> &vec, limb y,
+                                                 size_t start) noexcept {
+  size_t index = start;
+  limb carry = y;
+  bool overflow;
+  while (carry != 0 && index < vec.len()) {
+    vec[index] = scalar_add(vec[index], carry, overflow);
+    carry = limb(overflow);
+    index += 1;
+  }
+  if (carry != 0) {
+    SIMDJSON_FASTFLOAT_TRY(vec.try_push(carry));
+  }
+  return true;
+}

-/*!
-The format of the resulting decimal representation is similar to printf's %g
-format. Returns an iterator pointing past-the-end of the decimal representation.
-@note The input number must be finite, i.e. NaN's and Inf's are not supported.
-@note The buffer must be large enough.
-@note The result is NOT null-terminated.
-*/
-char *to_chars(char *first, const char *last, double value) {
-  static_cast<void>(last); // maybe unused - fix warning
-  bool negative = std::signbit(value);
-  if (negative) {
-    value = -value;
-    *first++ = '-';
+// add scalar value to bigint.
+template <uint16_t size>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 bool
+small_add(stackvec<size> &vec, limb y) noexcept {
+  return small_add_from(vec, y, 0);
+}
+
+// multiply bigint by scalar value.
+template <uint16_t size>
+inline SIMDJSON_FASTFLOAT_CONSTEXPR20 bool small_mul(stackvec<size> &vec,
+                                            limb y) noexcept {
+  limb carry = 0;
+  for (size_t index = 0; index < vec.len(); index++) {
+    vec[index] = scalar_mul(vec[index], y, carry);
+  }
+  if (carry != 0) {
+    SIMDJSON_FASTFLOAT_TRY(vec.try_push(carry));
   }
+  return true;
+}

-  if (value == 0) // +-0
-  {
-    *first++ = '0';
-    // Make it look like a floating-point number (#362, #378)
-    *first++ = '.';
-    *first++ = '0';
-    return first;
+// add bigint to bigint starting from index.
+// used in grade school multiplication
+template <uint16_t size>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 bool large_add_from(stackvec<size> &x, limb_span y,
+                                          size_t start) noexcept {
+  // the effective x buffer is from `xstart..x.len()`, so exit early
+  // if we can't get that current range.
+  if (x.len() < start || y.len() > x.len() - start) {
+    SIMDJSON_FASTFLOAT_TRY(x.try_resize(y.len() + start, 0));
   }
-  // Compute v = buffer * 10^decimal_exponent.
-  // The decimal digits are stored in the buffer, which needs to be interpreted
-  // as an unsigned decimal integer.
-  // len is the length of the buffer, i.e. the number of decimal digits.
-  int len = 0;
-  int decimal_exponent = 0;
-  dtoa_impl::grisu2(first, len, decimal_exponent, value);
-  // Format the buffer like printf("%.*g", prec, value)
-  constexpr int kMinExp = -4;
-  constexpr int kMaxExp = std::numeric_limits<double>::digits10;

-  return dtoa_impl::format_buffer(first, len, decimal_exponent, kMinExp,
-                                  kMaxExp);
+  bool carry = false;
+  for (size_t index = 0; index < y.len(); index++) {
+    limb xi = x[index + start];
+    limb yi = y[index];
+    bool c1 = false;
+    bool c2 = false;
+    xi = scalar_add(xi, yi, c1);
+    if (carry) {
+      xi = scalar_add(xi, 1, c2);
+    }
+    x[index + start] = xi;
+    carry = c1 | c2;
+  }
+
+  // handle overflow
+  if (carry) {
+    SIMDJSON_FASTFLOAT_TRY(small_add_from(x, 1, y.len() + start));
+  }
+  return true;
 }
-} // namespace internal
-} // namespace simdjson

-#endif // SIMDJSON_SRC_TO_CHARS_CPP
-/* end file to_chars.cpp */
-/* including from_chars.cpp: #include <from_chars.cpp> */
-/* begin file from_chars.cpp */
-#ifndef SIMDJSON_SRC_FROM_CHARS_CPP
-#define SIMDJSON_SRC_FROM_CHARS_CPP
+// add bigint to bigint.
+template <uint16_t size>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 bool
+large_add_from(stackvec<size> &x, limb_span y) noexcept {
+  return large_add_from(x, y, 0);
+}
+
+// grade-school multiplication algorithm
+template <uint16_t size>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 bool long_mul(stackvec<size> &x, limb_span y) noexcept {
+  limb_span xs = limb_span(x.data, x.len());
+  stackvec<size> z(xs);
+  limb_span zs = limb_span(z.data, z.len());
+
+  if (y.len() != 0) {
+    limb y0 = y[0];
+    SIMDJSON_FASTFLOAT_TRY(small_mul(x, y0));
+    for (size_t index = 1; index < y.len(); index++) {
+      limb yi = y[index];
+      stackvec<size> zi;
+      if (yi != 0) {
+        // re-use the same buffer throughout
+        zi.set_len(0);
+        SIMDJSON_FASTFLOAT_TRY(zi.try_extend(zs));
+        SIMDJSON_FASTFLOAT_TRY(small_mul(zi, yi));
+        limb_span zis = limb_span(zi.data, zi.len());
+        SIMDJSON_FASTFLOAT_TRY(large_add_from(x, zis, index));
+      }
+    }
+  }

-/* skipped duplicate #include <base.h> */
+  x.normalize();
+  return true;
+}

-#include <cstdint>
-#include <cstring>
-#include <limits>
+// grade-school multiplication algorithm
+template <uint16_t size>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 bool large_mul(stackvec<size> &x, limb_span y) noexcept {
+  if (y.len() == 1) {
+    SIMDJSON_FASTFLOAT_TRY(small_mul(x, y[0]));
+  } else {
+    SIMDJSON_FASTFLOAT_TRY(long_mul(x, y));
+  }
+  return true;
+}

-namespace simdjson {
-namespace internal {
+template <typename = void> struct pow5_tables {
+  static constexpr uint32_t large_step = 135;
+  static constexpr uint64_t small_power_of_5[] = {
+      1UL,
+      5UL,
+      25UL,
+      125UL,
+      625UL,
+      3125UL,
+      15625UL,
+      78125UL,
+      390625UL,
+      1953125UL,
+      9765625UL,
+      48828125UL,
+      244140625UL,
+      1220703125UL,
+      6103515625UL,
+      30517578125UL,
+      152587890625UL,
+      762939453125UL,
+      3814697265625UL,
+      19073486328125UL,
+      95367431640625UL,
+      476837158203125UL,
+      2384185791015625UL,
+      11920928955078125UL,
+      59604644775390625UL,
+      298023223876953125UL,
+      1490116119384765625UL,
+      7450580596923828125UL,
+  };
+#ifdef SIMDJSON_FASTFLOAT_64BIT_LIMB
+  constexpr static limb large_power_of_5[] = {
+      1414648277510068013UL, 9180637584431281687UL, 4539964771860779200UL,
+      10482974169319127550UL, 198276706040285095UL};
+#else
+  constexpr static limb large_power_of_5[] = {
+      4279965485U, 329373468U,  4020270615U, 2137533757U, 4287402176U,
+      1057042919U, 1071430142U, 2440757623U, 381945767U,  46164893U};
+#endif
+};

-/**
- * The code in the internal::from_chars function is meant to handle the floating-point number parsing
- * when we have more than 19 digits in the decimal mantissa. This should only be seen
- * in adversarial scenarios: we do not expect production systems to even produce
- * such floating-point numbers.
- *
- * The parser is based on work by Nigel Tao (at https://github.com/google/wuffs/)
- * who credits Ken Thompson for the design (via a reference to the Go source
- * code). See
- * https://github.com/google/wuffs/blob/aa46859ea40c72516deffa1b146121952d6dfd3b/internal/cgen/base/floatconv-submodule-data.c
- * https://github.com/google/wuffs/blob/46cd8105f47ca07ae2ba8e6a7818ef9c0df6c152/internal/cgen/base/floatconv-submodule-code.c
- * It is probably not very fast but it is a fallback that should almost never be
- * called in real life. Google Wuffs is published under APL 2.0.
- **/
+#if SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE

-namespace {
-constexpr uint32_t max_digits = 768;
-constexpr int32_t decimal_point_range = 2047;
-} // namespace
+template <typename T> constexpr uint32_t pow5_tables<T>::large_step;

-struct adjusted_mantissa {
-  uint64_t mantissa;
-  int power2;
-  adjusted_mantissa() : mantissa(0), power2(0) {}
-};
+template <typename T> constexpr uint64_t pow5_tables<T>::small_power_of_5[];

-struct decimal {
-  uint32_t num_digits;
-  int32_t decimal_point;
-  bool negative;
-  bool truncated;
-  uint8_t digits[max_digits];
-};
+template <typename T> constexpr limb pow5_tables<T>::large_power_of_5[];

-template <typename T> struct binary_format {
-  static constexpr int mantissa_explicit_bits();
-  static constexpr int minimum_exponent();
-  static constexpr int infinite_power();
-  static constexpr int sign_index();
-};
+#endif

-template <> constexpr int binary_format<double>::mantissa_explicit_bits() {
-  return 52;
-}
+// big integer type. implements a small subset of big integer
+// arithmetic, using simple algorithms since asymptotically
+// faster algorithms are slower for a small number of limbs.
+// all operations assume the big-integer is normalized.
+struct bigint : pow5_tables<> {
+  // storage of the limbs, in little-endian order.
+  stackvec<bigint_limbs> vec;

-template <> constexpr int binary_format<double>::minimum_exponent() {
-  return -1023;
-}
-template <> constexpr int binary_format<double>::infinite_power() {
-  return 0x7FF;
-}
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 bigint() : vec() {}

-template <> constexpr int binary_format<double>::sign_index() { return 63; }
+  bigint(bigint const &) = delete;
+  bigint &operator=(bigint const &) = delete;
+  bigint(bigint &&) = delete;
+  bigint &operator=(bigint &&other) = delete;

-bool is_integer(char c)  noexcept  { return (c >= '0' && c <= '9'); }
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 bigint(uint64_t value) : vec() {
+#ifdef SIMDJSON_FASTFLOAT_64BIT_LIMB
+    vec.push_unchecked(value);
+#else
+    vec.push_unchecked(uint32_t(value));
+    vec.push_unchecked(uint32_t(value >> 32));
+#endif
+    vec.normalize();
+  }

-// This should always succeed since it follows a call to parse_number.
-decimal parse_decimal(const char *&p) noexcept {
-  decimal answer;
-  answer.num_digits = 0;
-  answer.decimal_point = 0;
-  answer.truncated = false;
-  answer.negative = (*p == '-');
-  if ((*p == '-') || (*p == '+')) {
-    ++p;
+  // get the high 64 bits from the vector, and if bits were truncated.
+  // this is to get the significant digits for the float.
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 uint64_t hi64(bool &truncated) const noexcept {
+#ifdef SIMDJSON_FASTFLOAT_64BIT_LIMB
+    if (vec.len() == 0) {
+      return empty_hi64(truncated);
+    } else if (vec.len() == 1) {
+      return uint64_hi64(vec.rindex(0), truncated);
+    } else {
+      uint64_t result = uint64_hi64(vec.rindex(0), vec.rindex(1), truncated);
+      truncated |= vec.nonzero(2);
+      return result;
+    }
+#else
+    if (vec.len() == 0) {
+      return empty_hi64(truncated);
+    } else if (vec.len() == 1) {
+      return uint32_hi64(vec.rindex(0), truncated);
+    } else if (vec.len() == 2) {
+      return uint32_hi64(vec.rindex(0), vec.rindex(1), truncated);
+    } else {
+      uint64_t result =
+          uint32_hi64(vec.rindex(0), vec.rindex(1), vec.rindex(2), truncated);
+      truncated |= vec.nonzero(3);
+      return result;
+    }
+#endif
   }

-  while (*p == '0') {
-    ++p;
+  // compare two big integers, returning the large value.
+  // assumes both are normalized. if the return value is
+  // negative, other is larger, if the return value is
+  // positive, this is larger, otherwise they are equal.
+  // the limbs are stored in little-endian order, so we
+  // must compare the limbs in ever order.
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 int compare(bigint const &other) const noexcept {
+    if (vec.len() > other.vec.len()) {
+      return 1;
+    } else if (vec.len() < other.vec.len()) {
+      return -1;
+    } else {
+      for (size_t index = vec.len(); index > 0; index--) {
+        limb xi = vec[index - 1];
+        limb yi = other.vec[index - 1];
+        if (xi > yi) {
+          return 1;
+        } else if (xi < yi) {
+          return -1;
+        }
+      }
+      return 0;
+    }
   }
-  while (is_integer(*p)) {
-    if (answer.num_digits < max_digits) {
-      answer.digits[answer.num_digits] = uint8_t(*p - '0');
+
+  // shift left each limb n bits, carrying over to the new limb
+  // returns true if we were able to shift all the digits.
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 bool shl_bits(size_t n) noexcept {
+    // Internally, for each item, we shift left by n, and add the previous
+    // right shifted limb-bits.
+    // For example, we transform (for u8) shifted left 2, to:
+    //      b10100100 b01000010
+    //      b10 b10010001 b00001000
+    SIMDJSON_FASTFLOAT_DEBUG_ASSERT(n != 0);
+    SIMDJSON_FASTFLOAT_DEBUG_ASSERT(n < sizeof(limb) * 8);
+
+    size_t shl = n;
+    size_t shr = limb_bits - shl;
+    limb prev = 0;
+    for (size_t index = 0; index < vec.len(); index++) {
+      limb xi = vec[index];
+      vec[index] = (xi << shl) | (prev >> shr);
+      prev = xi;
     }
-    answer.num_digits++;
-    ++p;
+
+    limb carry = prev >> shr;
+    if (carry != 0) {
+      return vec.try_push(carry);
+    }
+    return true;
   }
-  if (*p == '.') {
-    ++p;
-    const char *first_after_period = p;
-    // if we have not yet encountered a zero, we have to skip it as well
-    if (answer.num_digits == 0) {
-      // skip zeros
-      while (*p == '0') {
-        ++p;
-      }
+
+  // move the limbs left by `n` limbs.
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 bool shl_limbs(size_t n) noexcept {
+    SIMDJSON_FASTFLOAT_DEBUG_ASSERT(n != 0);
+    if (n + vec.len() > vec.capacity()) {
+      return false;
+    } else if (!vec.is_empty()) {
+      // move limbs
+      limb *dst = vec.data + n;
+      limb const *src = vec.data;
+      std::copy_backward(src, src + vec.len(), dst + vec.len());
+      // fill in empty limbs
+      limb *first = vec.data;
+      limb *last = first + n;
+      ::std::fill(first, last, 0);
+      vec.set_len(n + vec.len());
+      return true;
+    } else {
+      return true;
     }
-    while (is_integer(*p)) {
-      if (answer.num_digits < max_digits) {
-        answer.digits[answer.num_digits] = uint8_t(*p - '0');
-      }
-      answer.num_digits++;
-      ++p;
+  }
+
+  // move the limbs left by `n` bits.
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 bool shl(size_t n) noexcept {
+    size_t rem = n % limb_bits;
+    size_t div = n / limb_bits;
+    if (rem != 0) {
+      SIMDJSON_FASTFLOAT_TRY(shl_bits(rem));
     }
-    answer.decimal_point = int32_t(first_after_period - p);
+    if (div != 0) {
+      SIMDJSON_FASTFLOAT_TRY(shl_limbs(div));
+    }
+    return true;
   }
-  if(answer.num_digits > 0) {
-    const char *preverse = p - 1;
-    int32_t trailing_zeros = 0;
-    while ((*preverse == '0') || (*preverse == '.')) {
-      if(*preverse == '0') { trailing_zeros++; };
-      --preverse;
+
+  // get the number of leading zeros in the bigint.
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 int ctlz() const noexcept {
+    if (vec.is_empty()) {
+      return 0;
+    } else {
+#ifdef SIMDJSON_FASTFLOAT_64BIT_LIMB
+      return leading_zeroes(vec.rindex(0));
+#else
+      // no use defining a specialized leading_zeroes for a 32-bit type.
+      uint64_t r0 = vec.rindex(0);
+      return leading_zeroes(r0 << 32);
+#endif
     }
-    answer.decimal_point += int32_t(answer.num_digits);
-    answer.num_digits -= uint32_t(trailing_zeros);
   }
-  if(answer.num_digits > max_digits ) {
-    answer.num_digits = max_digits;
-    answer.truncated = true;
+
+  // get the number of bits in the bigint.
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 int bit_length() const noexcept {
+    int lz = ctlz();
+    return int(limb_bits * vec.len()) - lz;
   }
-  if (('e' == *p) || ('E' == *p)) {
-    ++p;
-    bool neg_exp = false;
-    if ('-' == *p) {
-      neg_exp = true;
-      ++p;
-    } else if ('+' == *p) {
-      ++p;
+
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 bool mul(limb y) noexcept { return small_mul(vec, y); }
+
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 bool add(limb y) noexcept { return small_add(vec, y); }
+
+  // multiply as if by 2 raised to a power.
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 bool pow2(uint32_t exp) noexcept { return shl(exp); }
+
+  // multiply as if by 5 raised to a power.
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 bool pow5(uint32_t exp) noexcept {
+    // multiply by a power of 5
+    size_t large_length = sizeof(large_power_of_5) / sizeof(limb);
+    limb_span large = limb_span(large_power_of_5, large_length);
+    while (exp >= large_step) {
+      SIMDJSON_FASTFLOAT_TRY(large_mul(vec, large));
+      exp -= large_step;
     }
-    int32_t exp_number = 0; // exponential part
-    while (is_integer(*p)) {
-      uint8_t digit = uint8_t(*p - '0');
-      if (exp_number < 0x10000) {
-        exp_number = 10 * exp_number + digit;
-      }
-      ++p;
+#ifdef SIMDJSON_FASTFLOAT_64BIT_LIMB
+    uint32_t small_step = 27;
+    limb max_native = 7450580596923828125UL;
+#else
+    uint32_t small_step = 13;
+    limb max_native = 1220703125U;
+#endif
+    while (exp >= small_step) {
+      SIMDJSON_FASTFLOAT_TRY(small_mul(vec, max_native));
+      exp -= small_step;
+    }
+    if (exp != 0) {
+      // Work around clang bug https://godbolt.org/z/zedh7rrhc
+      // This is similar to https://github.com/llvm/llvm-project/issues/47746,
+      // except the workaround described there don't work here
+      SIMDJSON_FASTFLOAT_TRY(small_mul(vec, limb((static_cast<void>(small_power_of_5[0]),
+                                         small_power_of_5[exp]))));
     }
-    answer.decimal_point += (neg_exp ? -exp_number : exp_number);
+
+    return true;
   }
-  return answer;
+
+  // multiply as if by 10 raised to a power.
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 bool pow10(uint32_t exp) noexcept {
+    SIMDJSON_FASTFLOAT_TRY(pow5(exp));
+    return pow2(exp);
+  }
+};
+
+} // namespace simdjson_fast_float
+
+#endif
+
+#ifndef SIMDJSON_FASTFLOAT_DIGIT_COMPARISON_H
+#define SIMDJSON_FASTFLOAT_DIGIT_COMPARISON_H
+
+#include <cstdint>
+#include <cstring>
+#include <iterator>
+
+
+namespace simdjson_fast_float {
+
+// 1e0 to 1e19
+constexpr static uint64_t powers_of_ten_uint64[] = {1UL,
+                                                    10UL,
+                                                    100UL,
+                                                    1000UL,
+                                                    10000UL,
+                                                    100000UL,
+                                                    1000000UL,
+                                                    10000000UL,
+                                                    100000000UL,
+                                                    1000000000UL,
+                                                    10000000000UL,
+                                                    100000000000UL,
+                                                    1000000000000UL,
+                                                    10000000000000UL,
+                                                    100000000000000UL,
+                                                    1000000000000000UL,
+                                                    10000000000000000UL,
+                                                    100000000000000000UL,
+                                                    1000000000000000000UL,
+                                                    10000000000000000000UL};
+
+// calculate the exponent, in scientific notation, of the number.
+// this algorithm is not even close to optimized, but it has no practical
+// effect on performance: in order to have a faster algorithm, we'd need
+// to slow down performance for faster algorithms, and this is still fast.
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 int32_t
+scientific_exponent(uint64_t mantissa, int32_t exponent) noexcept {
+  while (mantissa >= 10000) {
+    mantissa /= 10000;
+    exponent += 4;
+  }
+  while (mantissa >= 100) {
+    mantissa /= 100;
+    exponent += 2;
+  }
+  while (mantissa >= 10) {
+    mantissa /= 10;
+    exponent += 1;
+  }
+  return exponent;
+}
+
+// this converts a native floating-point number to an extended-precision float.
+template <typename T>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 adjusted_mantissa
+to_extended(T value) noexcept {
+  using equiv_uint = equiv_uint_t<T>;
+  constexpr equiv_uint exponent_mask = binary_format<T>::exponent_mask();
+  constexpr equiv_uint mantissa_mask = binary_format<T>::mantissa_mask();
+  constexpr equiv_uint hidden_bit_mask = binary_format<T>::hidden_bit_mask();
+
+  adjusted_mantissa am;
+  int32_t bias = binary_format<T>::mantissa_explicit_bits() -
+                 binary_format<T>::minimum_exponent();
+  equiv_uint bits;
+#if SIMDJSON_FASTFLOAT_HAS_BIT_CAST
+  bits = std::bit_cast<equiv_uint>(value);
+#else
+  ::memcpy(&bits, &value, sizeof(T));
+#endif
+  if ((bits & exponent_mask) == 0) {
+    // denormal
+    am.power2 = 1 - bias;
+    am.mantissa = bits & mantissa_mask;
+  } else {
+    // normal
+    am.power2 = int32_t((bits & exponent_mask) >>
+                        binary_format<T>::mantissa_explicit_bits());
+    am.power2 -= bias;
+    am.mantissa = (bits & mantissa_mask) | hidden_bit_mask;
+  }
+
+  return am;
 }

-// This should always succeed since it follows a call to parse_number.
-// Will not read at or beyond the "end" pointer.
-decimal parse_decimal(const char *&p, const char * end) noexcept {
-  decimal answer;
-  answer.num_digits = 0;
-  answer.decimal_point = 0;
-  answer.truncated = false;
-  if(p == end) { return answer; } // should never happen
-  answer.negative = (*p == '-');
-  if ((*p == '-') || (*p == '+')) {
-    ++p;
+// get the extended precision value of the halfway point between b and b+u.
+// we are given a native float that represents b, so we need to adjust it
+// halfway between b and b+u.
+template <typename T>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 adjusted_mantissa
+to_extended_halfway(T value) noexcept {
+  adjusted_mantissa am = to_extended(value);
+  am.mantissa <<= 1;
+  am.mantissa += 1;
+  am.power2 -= 1;
+  return am;
+}
+
+// round an extended-precision float to the nearest machine float.
+template <typename T, typename callback>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 void round(adjusted_mantissa &am,
+                                                         callback cb) noexcept {
+  int32_t mantissa_shift = 64 - binary_format<T>::mantissa_explicit_bits() - 1;
+  if (-am.power2 >= mantissa_shift) {
+    // have a denormal float
+    int32_t shift = -am.power2 + 1;
+    cb(am, (shift < 64 ? shift : 64));
+    // check for round-up: if rounding-nearest carried us to the hidden bit.
+    am.power2 = (am.mantissa <
+                 (uint64_t(1) << binary_format<T>::mantissa_explicit_bits()))
+                    ? 0
+                    : 1;
+    return;
   }

-  while ((p != end) && (*p == '0')) {
-    ++p;
+  // have a normal float, use the default shift.
+  cb(am, mantissa_shift);
+
+  // check for carry
+  if (am.mantissa >=
+      (uint64_t(2) << binary_format<T>::mantissa_explicit_bits())) {
+    am.mantissa = (uint64_t(1) << binary_format<T>::mantissa_explicit_bits());
+    am.power2++;
+  }
+
+  // check for infinite: we could have carried to an infinite power
+  am.mantissa &= ~(uint64_t(1) << binary_format<T>::mantissa_explicit_bits());
+  if (am.power2 >= binary_format<T>::infinite_power()) {
+    am.power2 = binary_format<T>::infinite_power();
+    am.mantissa = 0;
+  }
+}
+
+template <typename callback>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 void
+round_nearest_tie_even(adjusted_mantissa &am, int32_t shift,
+                       callback cb) noexcept {
+  uint64_t const mask = (shift == 64) ? UINT64_MAX : (uint64_t(1) << shift) - 1;
+  uint64_t const halfway = (shift == 0) ? 0 : uint64_t(1) << (shift - 1);
+  uint64_t truncated_bits = am.mantissa & mask;
+  bool is_above = truncated_bits > halfway;
+  bool is_halfway = truncated_bits == halfway;
+
+  // shift digits into position
+  if (shift == 64) {
+    am.mantissa = 0;
+  } else {
+    am.mantissa >>= shift;
+  }
+  am.power2 += shift;
+
+  bool is_odd = (am.mantissa & 1) == 1;
+  am.mantissa += uint64_t(cb(is_odd, is_halfway, is_above));
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 void
+round_down(adjusted_mantissa &am, int32_t shift) noexcept {
+  if (shift == 64) {
+    am.mantissa = 0;
+  } else {
+    am.mantissa >>= shift;
   }
-  while ((p != end) && is_integer(*p)) {
-    if (answer.num_digits < max_digits) {
-      answer.digits[answer.num_digits] = uint8_t(*p - '0');
+  am.power2 += shift;
+}
+
+template <typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 void
+skip_zeros(UC const *&first, UC const *last) noexcept {
+  uint64_t val;
+  while (!cpp20_and_in_constexpr() &&
+         std::distance(first, last) >= int_cmp_len<UC>()) {
+    ::memcpy(&val, first, sizeof(uint64_t));
+    if (val != int_cmp_zeros<UC>()) {
+      break;
     }
-    answer.num_digits++;
-    ++p;
+    first += int_cmp_len<UC>();
   }
-  if ((p != end) && (*p == '.')) {
-    ++p;
-    if(p == end) { return answer; } // should never happen
-    const char *first_after_period = p;
-    // if we have not yet encountered a zero, we have to skip it as well
-    if (answer.num_digits == 0) {
-      // skip zeros
-      while (*p == '0') {
-        ++p;
-      }
+  while (first != last) {
+    if (*first != UC('0')) {
+      break;
     }
-    while ((p != end) && is_integer(*p)) {
-      if (answer.num_digits < max_digits) {
-        answer.digits[answer.num_digits] = uint8_t(*p - '0');
-      }
-      answer.num_digits++;
-      ++p;
+    first++;
+  }
+}
+
+// determine if any non-zero digits were truncated.
+// all characters must be valid digits.
+template <typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 bool
+is_truncated(UC const *first, UC const *last) noexcept {
+  // do 8-bit optimizations, can just compare to 8 literal 0s.
+  uint64_t val;
+  while (!cpp20_and_in_constexpr() &&
+         std::distance(first, last) >= int_cmp_len<UC>()) {
+    ::memcpy(&val, first, sizeof(uint64_t));
+    if (val != int_cmp_zeros<UC>()) {
+      return true;
     }
-    answer.decimal_point = int32_t(first_after_period - p);
+    first += int_cmp_len<UC>();
   }
-  if(answer.num_digits > 0) {
-    const char *preverse = p - 1;
-    int32_t trailing_zeros = 0;
-    while ((*preverse == '0') || (*preverse == '.')) {
-      if(*preverse == '0') { trailing_zeros++; };
-      --preverse;
+  while (first != last) {
+    if (*first != UC('0')) {
+      return true;
     }
-    answer.decimal_point += int32_t(answer.num_digits);
-    answer.num_digits -= uint32_t(trailing_zeros);
+    ++first;
   }
-  if(answer.num_digits > max_digits ) {
-    answer.num_digits = max_digits;
-    answer.truncated = true;
+  return false;
+}
+
+template <typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 bool
+is_truncated(span<UC const> s) noexcept {
+  return is_truncated(s.ptr, s.ptr + s.len());
+}
+
+template <typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 void
+parse_eight_digits(UC const *&p, limb &value, size_t &counter,
+                   size_t &count) noexcept {
+  value = value * 100000000 + parse_eight_digits_unrolled(p);
+  p += 8;
+  counter += 8;
+  count += 8;
+}
+
+template <typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 void
+parse_one_digit(UC const *&p, limb &value, size_t &counter,
+                size_t &count) noexcept {
+  value = value * 10 + limb(*p - UC('0'));
+  p++;
+  counter++;
+  count++;
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 void
+add_native(bigint &big, limb power, limb value) noexcept {
+  big.mul(power);
+  big.add(value);
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 void
+round_up_bigint(bigint &big, size_t &count) noexcept {
+  // need to round-up the digits, but need to avoid rounding
+  // ....9999 to ...10000, which could cause a false halfway point.
+  add_native(big, 10, 1);
+  count++;
+}
+
+// parse the significant digits into a big integer
+template <typename UC>
+inline SIMDJSON_FASTFLOAT_CONSTEXPR20 void
+parse_mantissa(bigint &result, parsed_number_string_t<UC> &num,
+               size_t max_digits, size_t &digits) noexcept {
+  // try to minimize the number of big integer and scalar multiplication.
+  // therefore, try to parse 8 digits at a time, and multiply by the largest
+  // scalar value (9 or 19 digits) for each step.
+  size_t counter = 0;
+  digits = 0;
+  limb value = 0;
+#ifdef SIMDJSON_FASTFLOAT_64BIT_LIMB
+  size_t step = 19;
+#else
+  size_t step = 9;
+#endif
+
+  // process all integer digits.
+  UC const *p = num.integer.ptr;
+  UC const *pend = p + num.integer.len();
+  skip_zeros(p, pend);
+  // process all digits, in increments of step per loop
+  while (p != pend) {
+    while ((std::distance(p, pend) >= 8) && (step - counter >= 8) &&
+           (max_digits - digits >= 8)) {
+      parse_eight_digits(p, value, counter, digits);
+    }
+    while (counter < step && p != pend && digits < max_digits) {
+      parse_one_digit(p, value, counter, digits);
+    }
+    if (digits == max_digits) {
+      // add the temporary value, then check if we've truncated any digits
+      add_native(result, limb(powers_of_ten_uint64[counter]), value);
+      bool truncated = is_truncated(p, pend);
+      if (num.fraction.ptr != nullptr) {
+        truncated |= is_truncated(num.fraction);
+      }
+      if (truncated) {
+        round_up_bigint(result, digits);
+      }
+      return;
+    } else {
+      add_native(result, limb(powers_of_ten_uint64[counter]), value);
+      counter = 0;
+      value = 0;
+    }
   }
-  if ((p != end) && (('e' == *p) || ('E' == *p))) {
-    ++p;
-    if(p == end) { return answer; } // should never happen
-    bool neg_exp = false;
-    if ('-' == *p) {
-      neg_exp = true;
-      ++p;
-    } else if ('+' == *p) {
-      ++p;
+
+  // add our fraction digits, if they're available.
+  if (num.fraction.ptr != nullptr) {
+    p = num.fraction.ptr;
+    pend = p + num.fraction.len();
+    if (digits == 0) {
+      skip_zeros(p, pend);
     }
-    int32_t exp_number = 0; // exponential part
-    while ((p != end) && is_integer(*p)) {
-      uint8_t digit = uint8_t(*p - '0');
-      if (exp_number < 0x10000) {
-        exp_number = 10 * exp_number + digit;
+    // process all digits, in increments of step per loop
+    while (p != pend) {
+      while ((std::distance(p, pend) >= 8) && (step - counter >= 8) &&
+             (max_digits - digits >= 8)) {
+        parse_eight_digits(p, value, counter, digits);
+      }
+      while (counter < step && p != pend && digits < max_digits) {
+        parse_one_digit(p, value, counter, digits);
+      }
+      if (digits == max_digits) {
+        // add the temporary value, then check if we've truncated any digits
+        add_native(result, limb(powers_of_ten_uint64[counter]), value);
+        bool truncated = is_truncated(p, pend);
+        if (truncated) {
+          round_up_bigint(result, digits);
+        }
+        return;
+      } else {
+        add_native(result, limb(powers_of_ten_uint64[counter]), value);
+        counter = 0;
+        value = 0;
       }
-      ++p;
     }
-    answer.decimal_point += (neg_exp ? -exp_number : exp_number);
   }
+
+  if (counter != 0) {
+    add_native(result, limb(powers_of_ten_uint64[counter]), value);
+  }
+}
+
+template <typename T>
+inline SIMDJSON_FASTFLOAT_CONSTEXPR20 adjusted_mantissa
+positive_digit_comp(bigint &bigmant, int32_t exponent) noexcept {
+  SIMDJSON_FASTFLOAT_ASSERT(bigmant.pow10(uint32_t(exponent)));
+  adjusted_mantissa answer;
+  bool truncated;
+  answer.mantissa = bigmant.hi64(truncated);
+  int bias = binary_format<T>::mantissa_explicit_bits() -
+             binary_format<T>::minimum_exponent();
+  answer.power2 = bigmant.bit_length() - 64 + bias;
+
+  round<T>(answer, [truncated](adjusted_mantissa &a, int32_t shift) {
+    round_nearest_tie_even(
+        a, shift,
+        [truncated](bool is_odd, bool is_halfway, bool is_above) -> bool {
+          return is_above || (is_halfway && truncated) ||
+                 (is_odd && is_halfway);
+        });
+  });
+
   return answer;
 }

-namespace {
+// the scaling here is quite simple: we have, for the real digits `m * 10^e`,
+// and for the theoretical digits `n * 2^f`. Since `e` is always negative,
+// to scale them identically, we do `n * 2^f * 5^-f`, so we now have `m * 2^e`.
+// we then need to scale by `2^(f- e)`, and then the two significant digits
+// are of the same magnitude.
+template <typename T>
+inline SIMDJSON_FASTFLOAT_CONSTEXPR20 adjusted_mantissa negative_digit_comp(
+    bigint &bigmant, adjusted_mantissa am, int32_t exponent) noexcept {
+  bigint &real_digits = bigmant;
+  int32_t real_exp = exponent;
+
+  // get the value of `b`, rounded down, and get a bigint representation of b+h
+  adjusted_mantissa am_b = am;
+  // gcc7 buf: use a lambda to remove the noexcept qualifier bug with
+  // -Wnoexcept-type.
+  round<T>(am_b,
+           [](adjusted_mantissa &a, int32_t shift) { round_down(a, shift); });
+  T b;
+  to_float(false, am_b, b);
+  adjusted_mantissa theor = to_extended_halfway(b);
+  bigint theor_digits(theor.mantissa);
+  int32_t theor_exp = theor.power2;
+
+  // scale real digits and theor digits to be same power.
+  int32_t pow2_exp = theor_exp - real_exp;
+  uint32_t pow5_exp = uint32_t(-real_exp);
+  if (pow5_exp != 0) {
+    SIMDJSON_FASTFLOAT_ASSERT(theor_digits.pow5(pow5_exp));
+  }
+  if (pow2_exp > 0) {
+    SIMDJSON_FASTFLOAT_ASSERT(theor_digits.pow2(uint32_t(pow2_exp)));
+  } else if (pow2_exp < 0) {
+    SIMDJSON_FASTFLOAT_ASSERT(real_digits.pow2(uint32_t(-pow2_exp)));
+  }
+
+  // compare digits, and use it to direct rounding
+  int ord = real_digits.compare(theor_digits);
+  adjusted_mantissa answer = am;
+  round<T>(answer, [ord](adjusted_mantissa &a, int32_t shift) {
+    round_nearest_tie_even(
+        a, shift, [ord](bool is_odd, bool _, bool __) -> bool {
+          static_cast<void>(_);  // not needed, since we've done our comparison
+          static_cast<void>(__); // not needed, since we've done our comparison
+          if (ord > 0) {
+            return true;
+          } else if (ord < 0) {
+            return false;
+          } else {
+            return is_odd;
+          }
+        });
+  });
+
+  return answer;
+}

-// remove all final zeroes
-inline void trim(decimal &h) {
-  while ((h.num_digits > 0) && (h.digits[h.num_digits - 1] == 0)) {
-    h.num_digits--;
+// parse the significant digits as a big integer to unambiguously round
+// the significant digits. here, we are trying to determine how to round
+// an extended float representation close to `b+h`, halfway between `b`
+// (the float rounded-down) and `b+u`, the next positive float. this
+// algorithm is always correct, and uses one of two approaches. when
+// the exponent is positive relative to the significant digits (such as
+// 1234), we create a big-integer representation, get the high 64-bits,
+// determine if any lower bits are truncated, and use that to direct
+// rounding. in case of a negative exponent relative to the significant
+// digits (such as 1.2345), we create a theoretical representation of
+// `b` as a big-integer type, scaled to the same binary exponent as
+// the actual digits. we then compare the big integer representations
+// of both, and use that to direct rounding.
+template <typename T, typename UC>
+inline SIMDJSON_FASTFLOAT_CONSTEXPR20 adjusted_mantissa
+digit_comp(parsed_number_string_t<UC> &num, adjusted_mantissa am) noexcept {
+  // remove the invalid exponent bias
+  am.power2 -= invalid_am_bias;
+
+  int32_t sci_exp =
+      scientific_exponent(num.mantissa, static_cast<int32_t>(num.exponent));
+  size_t max_digits = binary_format<T>::max_digits();
+  size_t digits = 0;
+  bigint bigmant;
+  parse_mantissa(bigmant, num, max_digits, digits);
+  // can't underflow, since digits is at most max_digits.
+  int32_t exponent = sci_exp + 1 - int32_t(digits);
+  if (exponent >= 0) {
+    return positive_digit_comp<T>(bigmant, exponent);
+  } else {
+    return negative_digit_comp<T>(bigmant, am, exponent);
   }
 }

-uint32_t number_of_digits_decimal_left_shift(decimal &h, uint32_t shift) {
-  shift &= 63;
-  const static uint16_t number_of_digits_decimal_left_shift_table[65] = {
-      0x0000, 0x0800, 0x0801, 0x0803, 0x1006, 0x1009, 0x100D, 0x1812, 0x1817,
-      0x181D, 0x2024, 0x202B, 0x2033, 0x203C, 0x2846, 0x2850, 0x285B, 0x3067,
-      0x3073, 0x3080, 0x388E, 0x389C, 0x38AB, 0x38BB, 0x40CC, 0x40DD, 0x40EF,
-      0x4902, 0x4915, 0x4929, 0x513E, 0x5153, 0x5169, 0x5180, 0x5998, 0x59B0,
-      0x59C9, 0x61E3, 0x61FD, 0x6218, 0x6A34, 0x6A50, 0x6A6D, 0x6A8B, 0x72AA,
-      0x72C9, 0x72E9, 0x7B0A, 0x7B2B, 0x7B4D, 0x8370, 0x8393, 0x83B7, 0x83DC,
-      0x8C02, 0x8C28, 0x8C4F, 0x9477, 0x949F, 0x94C8, 0x9CF2, 0x051C, 0x051C,
-      0x051C, 0x051C,
-  };
-  uint32_t x_a = number_of_digits_decimal_left_shift_table[shift];
-  uint32_t x_b = number_of_digits_decimal_left_shift_table[shift + 1];
-  uint32_t num_new_digits = x_a >> 11;
-  uint32_t pow5_a = 0x7FF & x_a;
-  uint32_t pow5_b = 0x7FF & x_b;
-  const static uint8_t
-      number_of_digits_decimal_left_shift_table_powers_of_5[0x051C] = {
-          5, 2, 5, 1, 2, 5, 6, 2, 5, 3, 1, 2, 5, 1, 5, 6, 2, 5, 7, 8, 1, 2, 5,
-          3, 9, 0, 6, 2, 5, 1, 9, 5, 3, 1, 2, 5, 9, 7, 6, 5, 6, 2, 5, 4, 8, 8,
-          2, 8, 1, 2, 5, 2, 4, 4, 1, 4, 0, 6, 2, 5, 1, 2, 2, 0, 7, 0, 3, 1, 2,
-          5, 6, 1, 0, 3, 5, 1, 5, 6, 2, 5, 3, 0, 5, 1, 7, 5, 7, 8, 1, 2, 5, 1,
-          5, 2, 5, 8, 7, 8, 9, 0, 6, 2, 5, 7, 6, 2, 9, 3, 9, 4, 5, 3, 1, 2, 5,
-          3, 8, 1, 4, 6, 9, 7, 2, 6, 5, 6, 2, 5, 1, 9, 0, 7, 3, 4, 8, 6, 3, 2,
-          8, 1, 2, 5, 9, 5, 3, 6, 7, 4, 3, 1, 6, 4, 0, 6, 2, 5, 4, 7, 6, 8, 3,
-          7, 1, 5, 8, 2, 0, 3, 1, 2, 5, 2, 3, 8, 4, 1, 8, 5, 7, 9, 1, 0, 1, 5,
-          6, 2, 5, 1, 1, 9, 2, 0, 9, 2, 8, 9, 5, 5, 0, 7, 8, 1, 2, 5, 5, 9, 6,
-          0, 4, 6, 4, 4, 7, 7, 5, 3, 9, 0, 6, 2, 5, 2, 9, 8, 0, 2, 3, 2, 2, 3,
-          8, 7, 6, 9, 5, 3, 1, 2, 5, 1, 4, 9, 0, 1, 1, 6, 1, 1, 9, 3, 8, 4, 7,
-          6, 5, 6, 2, 5, 7, 4, 5, 0, 5, 8, 0, 5, 9, 6, 9, 2, 3, 8, 2, 8, 1, 2,
-          5, 3, 7, 2, 5, 2, 9, 0, 2, 9, 8, 4, 6, 1, 9, 1, 4, 0, 6, 2, 5, 1, 8,
-          6, 2, 6, 4, 5, 1, 4, 9, 2, 3, 0, 9, 5, 7, 0, 3, 1, 2, 5, 9, 3, 1, 3,
-          2, 2, 5, 7, 4, 6, 1, 5, 4, 7, 8, 5, 1, 5, 6, 2, 5, 4, 6, 5, 6, 6, 1,
-          2, 8, 7, 3, 0, 7, 7, 3, 9, 2, 5, 7, 8, 1, 2, 5, 2, 3, 2, 8, 3, 0, 6,
-          4, 3, 6, 5, 3, 8, 6, 9, 6, 2, 8, 9, 0, 6, 2, 5, 1, 1, 6, 4, 1, 5, 3,
-          2, 1, 8, 2, 6, 9, 3, 4, 8, 1, 4, 4, 5, 3, 1, 2, 5, 5, 8, 2, 0, 7, 6,
-          6, 0, 9, 1, 3, 4, 6, 7, 4, 0, 7, 2, 2, 6, 5, 6, 2, 5, 2, 9, 1, 0, 3,
-          8, 3, 0, 4, 5, 6, 7, 3, 3, 7, 0, 3, 6, 1, 3, 2, 8, 1, 2, 5, 1, 4, 5,
-          5, 1, 9, 1, 5, 2, 2, 8, 3, 6, 6, 8, 5, 1, 8, 0, 6, 6, 4, 0, 6, 2, 5,
-          7, 2, 7, 5, 9, 5, 7, 6, 1, 4, 1, 8, 3, 4, 2, 5, 9, 0, 3, 3, 2, 0, 3,
-          1, 2, 5, 3, 6, 3, 7, 9, 7, 8, 8, 0, 7, 0, 9, 1, 7, 1, 2, 9, 5, 1, 6,
-          6, 0, 1, 5, 6, 2, 5, 1, 8, 1, 8, 9, 8, 9, 4, 0, 3, 5, 4, 5, 8, 5, 6,
-          4, 7, 5, 8, 3, 0, 0, 7, 8, 1, 2, 5, 9, 0, 9, 4, 9, 4, 7, 0, 1, 7, 7,
-          2, 9, 2, 8, 2, 3, 7, 9, 1, 5, 0, 3, 9, 0, 6, 2, 5, 4, 5, 4, 7, 4, 7,
-          3, 5, 0, 8, 8, 6, 4, 6, 4, 1, 1, 8, 9, 5, 7, 5, 1, 9, 5, 3, 1, 2, 5,
-          2, 2, 7, 3, 7, 3, 6, 7, 5, 4, 4, 3, 2, 3, 2, 0, 5, 9, 4, 7, 8, 7, 5,
-          9, 7, 6, 5, 6, 2, 5, 1, 1, 3, 6, 8, 6, 8, 3, 7, 7, 2, 1, 6, 1, 6, 0,
-          2, 9, 7, 3, 9, 3, 7, 9, 8, 8, 2, 8, 1, 2, 5, 5, 6, 8, 4, 3, 4, 1, 8,
-          8, 6, 0, 8, 0, 8, 0, 1, 4, 8, 6, 9, 6, 8, 9, 9, 4, 1, 4, 0, 6, 2, 5,
-          2, 8, 4, 2, 1, 7, 0, 9, 4, 3, 0, 4, 0, 4, 0, 0, 7, 4, 3, 4, 8, 4, 4,
-          9, 7, 0, 7, 0, 3, 1, 2, 5, 1, 4, 2, 1, 0, 8, 5, 4, 7, 1, 5, 2, 0, 2,
-          0, 0, 3, 7, 1, 7, 4, 2, 2, 4, 8, 5, 3, 5, 1, 5, 6, 2, 5, 7, 1, 0, 5,
-          4, 2, 7, 3, 5, 7, 6, 0, 1, 0, 0, 1, 8, 5, 8, 7, 1, 1, 2, 4, 2, 6, 7,
-          5, 7, 8, 1, 2, 5, 3, 5, 5, 2, 7, 1, 3, 6, 7, 8, 8, 0, 0, 5, 0, 0, 9,
-          2, 9, 3, 5, 5, 6, 2, 1, 3, 3, 7, 8, 9, 0, 6, 2, 5, 1, 7, 7, 6, 3, 5,
-          6, 8, 3, 9, 4, 0, 0, 2, 5, 0, 4, 6, 4, 6, 7, 7, 8, 1, 0, 6, 6, 8, 9,
-          4, 5, 3, 1, 2, 5, 8, 8, 8, 1, 7, 8, 4, 1, 9, 7, 0, 0, 1, 2, 5, 2, 3,
-          2, 3, 3, 8, 9, 0, 5, 3, 3, 4, 4, 7, 2, 6, 5, 6, 2, 5, 4, 4, 4, 0, 8,
-          9, 2, 0, 9, 8, 5, 0, 0, 6, 2, 6, 1, 6, 1, 6, 9, 4, 5, 2, 6, 6, 7, 2,
-          3, 6, 3, 2, 8, 1, 2, 5, 2, 2, 2, 0, 4, 4, 6, 0, 4, 9, 2, 5, 0, 3, 1,
-          3, 0, 8, 0, 8, 4, 7, 2, 6, 3, 3, 3, 6, 1, 8, 1, 6, 4, 0, 6, 2, 5, 1,
-          1, 1, 0, 2, 2, 3, 0, 2, 4, 6, 2, 5, 1, 5, 6, 5, 4, 0, 4, 2, 3, 6, 3,
-          1, 6, 6, 8, 0, 9, 0, 8, 2, 0, 3, 1, 2, 5, 5, 5, 5, 1, 1, 1, 5, 1, 2,
-          3, 1, 2, 5, 7, 8, 2, 7, 0, 2, 1, 1, 8, 1, 5, 8, 3, 4, 0, 4, 5, 4, 1,
-          0, 1, 5, 6, 2, 5, 2, 7, 7, 5, 5, 5, 7, 5, 6, 1, 5, 6, 2, 8, 9, 1, 3,
-          5, 1, 0, 5, 9, 0, 7, 9, 1, 7, 0, 2, 2, 7, 0, 5, 0, 7, 8, 1, 2, 5, 1,
-          3, 8, 7, 7, 7, 8, 7, 8, 0, 7, 8, 1, 4, 4, 5, 6, 7, 5, 5, 2, 9, 5, 3,
-          9, 5, 8, 5, 1, 1, 3, 5, 2, 5, 3, 9, 0, 6, 2, 5, 6, 9, 3, 8, 8, 9, 3,
-          9, 0, 3, 9, 0, 7, 2, 2, 8, 3, 7, 7, 6, 4, 7, 6, 9, 7, 9, 2, 5, 5, 6,
-          7, 6, 2, 6, 9, 5, 3, 1, 2, 5, 3, 4, 6, 9, 4, 4, 6, 9, 5, 1, 9, 5, 3,
-          6, 1, 4, 1, 8, 8, 8, 2, 3, 8, 4, 8, 9, 6, 2, 7, 8, 3, 8, 1, 3, 4, 7,
-          6, 5, 6, 2, 5, 1, 7, 3, 4, 7, 2, 3, 4, 7, 5, 9, 7, 6, 8, 0, 7, 0, 9,
-          4, 4, 1, 1, 9, 2, 4, 4, 8, 1, 3, 9, 1, 9, 0, 6, 7, 3, 8, 2, 8, 1, 2,
-          5, 8, 6, 7, 3, 6, 1, 7, 3, 7, 9, 8, 8, 4, 0, 3, 5, 4, 7, 2, 0, 5, 9,
-          6, 2, 2, 4, 0, 6, 9, 5, 9, 5, 3, 3, 6, 9, 1, 4, 0, 6, 2, 5,
-      };
-  const uint8_t *pow5 =
-      &number_of_digits_decimal_left_shift_table_powers_of_5[pow5_a];
-  uint32_t i = 0;
-  uint32_t n = pow5_b - pow5_a;
-  for (; i < n; i++) {
-    if (i >= h.num_digits) {
-      return num_new_digits - 1;
-    } else if (h.digits[i] == pow5[i]) {
-      continue;
-    } else if (h.digits[i] < pow5[i]) {
-      return num_new_digits - 1;
-    } else {
-      return num_new_digits;
+} // namespace simdjson_fast_float
+
+#endif
+
+#ifndef SIMDJSON_FASTFLOAT_PARSE_NUMBER_H
+#define SIMDJSON_FASTFLOAT_PARSE_NUMBER_H
+
+
+#include <cmath>
+#include <cstring>
+#include <limits>
+#include <system_error>
+
+namespace simdjson_fast_float {
+
+namespace detail {
+/**
+ * Special case +inf, -inf, nan, infinity, -infinity.
+ * The case comparisons could be made much faster given that we know that the
+ * strings a null-free and fixed.
+ **/
+template <typename T, typename UC>
+from_chars_result_t<UC>
+    SIMDJSON_FASTFLOAT_CONSTEXPR14 parse_infnan(UC const *first, UC const *last,
+                                       T &value, chars_format fmt) noexcept {
+  from_chars_result_t<UC> answer{};
+  answer.ptr = first;
+  answer.ec = std::errc(); // be optimistic
+  // assume first < last, so dereference without checks;
+  bool const minusSign = (*first == UC('-'));
+  // C++17 20.19.3.(7.1) explicitly forbids '+' sign here
+  if ((*first == UC('-')) ||
+      (uint64_t(fmt & chars_format::allow_leading_plus) &&
+       (*first == UC('+')))) {
+    ++first;
+  }
+  if (last - first >= 3) {
+    if (simdjson_fastfloat_strncasecmp3(first, str_const_nan<UC>())) {
+      answer.ptr = (first += 3);
+      value = minusSign ? -std::numeric_limits<T>::quiet_NaN()
+                        : std::numeric_limits<T>::quiet_NaN();
+      // Check for possible nan(n-char-seq-opt), C++17 20.19.3.7,
+      // C11 7.20.1.3.3. At least MSVC produces nan(ind) and nan(snan).
+      if (first != last && *first == UC('(')) {
+        for (UC const *ptr = first + 1; ptr != last; ++ptr) {
+          if (*ptr == UC(')')) {
+            answer.ptr = ptr + 1; // valid nan(n-char-seq-opt)
+            break;
+          } else if (!((UC('a') <= *ptr && *ptr <= UC('z')) ||
+                       (UC('A') <= *ptr && *ptr <= UC('Z')) ||
+                       (UC('0') <= *ptr && *ptr <= UC('9')) || *ptr == UC('_')))
+            break; // forbidden char, not nan(n-char-seq-opt)
+        }
+      }
+      return answer;
+    }
+    if (simdjson_fastfloat_strncasecmp3(first, str_const_inf<UC>())) {
+      if ((last - first >= 8) &&
+          simdjson_fastfloat_strncasecmp5(first + 3, str_const_inf<UC>() + 3)) {
+        answer.ptr = first + 8;
+      } else {
+        answer.ptr = first + 3;
+      }
+      value = minusSign ? -std::numeric_limits<T>::infinity()
+                        : std::numeric_limits<T>::infinity();
+      return answer;
     }
   }
-  return num_new_digits;
+  answer.ec = std::errc::invalid_argument;
+  return answer;
+}
+
+/**
+ * Returns true if the floating-pointing rounding mode is to 'nearest'.
+ * It is the default on most system. This function is meant to be inexpensive.
+ * Credit : @mwalcott3
+ */
+simdjson_fastfloat_really_inline bool rounds_to_nearest() noexcept {
+  // https://lemire.me/blog/2020/06/26/gcc-not-nearest/
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+  return false;
+#endif
+  // See
+  // A fast function to check your floating-point rounding mode
+  // https://lemire.me/blog/2022/11/16/a-fast-function-to-check-your-floating-point-rounding-mode/
+  //
+  // This function is meant to be equivalent to :
+  // prior: #include <cfenv>
+  //  return fegetround() == FE_TONEAREST;
+  // However, it is expected to be much faster than the fegetround()
+  // function call.
+  //
+  // The volatile keyword prevents the compiler from computing the function
+  // at compile-time.
+  // There might be other ways to prevent compile-time optimizations (e.g.,
+  // asm). The value does not need to be std::numeric_limits<float>::min(), any
+  // small value so that 1 + x should round to 1 would do (after accounting for
+  // excess precision, as in 387 instructions).
+  static float volatile fmin = (std::numeric_limits<float>::min)();
+  float fmini = fmin; // we copy it so that it gets loaded at most once.
+//
+// Explanation:
+// Only when fegetround() == FE_TONEAREST do we have that
+// fmin + 1.0f == 1.0f - fmin.
+//
+// FE_UPWARD:
+//  fmin + 1.0f > 1
+//  1.0f - fmin == 1
+//
+// FE_DOWNWARD or  FE_TOWARDZERO:
+//  fmin + 1.0f == 1
+//  1.0f - fmin < 1
+//
+// Note: This may fail to be accurate if fast-math has been
+// enabled, as rounding conventions may not apply.
+#ifdef SIMDJSON_FASTFLOAT_VISUAL_STUDIO
+#pragma warning(push)
+//  todo: is there a VS warning?
+//  see
+//  https://stackoverflow.com/questions/46079446/is-there-a-warning-for-floating-point-equality-checking-in-visual-studio-2013
+#elif defined(__clang__)
+#pragma clang diagnostic push
+#pragma clang diagnostic ignored "-Wfloat-equal"
+#elif defined(__GNUC__)
+#pragma GCC diagnostic push
+#pragma GCC diagnostic ignored "-Wfloat-equal"
+#endif
+  return (fmini + 1.0f == 1.0f - fmini);
+#ifdef SIMDJSON_FASTFLOAT_VISUAL_STUDIO
+#pragma warning(pop)
+#elif defined(__clang__)
+#pragma clang diagnostic pop
+#elif defined(__GNUC__)
+#pragma GCC diagnostic pop
+#endif
 }

-} // end of anonymous namespace
+} // namespace detail

-uint64_t round(decimal &h) {
-  if ((h.num_digits == 0) || (h.decimal_point < 0)) {
-    return 0;
-  } else if (h.decimal_point > 18) {
-    return UINT64_MAX;
-  }
-  // at this point, we know that h.decimal_point >= 0
-  uint32_t dp = uint32_t(h.decimal_point);
-  uint64_t n = 0;
-  for (uint32_t i = 0; i < dp; i++) {
-    n = (10 * n) + ((i < h.num_digits) ? h.digits[i] : 0);
+template <typename T> struct from_chars_caller {
+  template <typename UC>
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 static from_chars_result_t<UC>
+  call(UC const *first, UC const *last, T &value,
+       parse_options_t<UC> options) noexcept {
+    return from_chars_advanced(first, last, value, options);
   }
-  bool round_up = false;
-  if (dp < h.num_digits) {
-    round_up = h.digits[dp] >= 5; // normally, we round up
-    // but we may need to round to even!
-    if ((h.digits[dp] == 5) && (dp + 1 == h.num_digits)) {
-      round_up = h.truncated || ((dp > 0) && (1 & h.digits[dp - 1]));
-    }
+};
+
+#ifdef __STDCPP_FLOAT32_T__
+template <> struct from_chars_caller<std::float32_t> {
+  template <typename UC>
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 static from_chars_result_t<UC>
+  call(UC const *first, UC const *last, std::float32_t &value,
+       parse_options_t<UC> options) noexcept {
+    // if std::float32_t is defined, and we are in C++23 mode; macro set for
+    // float32; set value to float due to equivalence between float and
+    // float32_t
+    float val = 0.0f;
+    auto ret = from_chars_advanced(first, last, val, options);
+    value = val;
+    return ret;
   }
-  if (round_up) {
-    n++;
+};
+#endif
+
+#ifdef __STDCPP_FLOAT64_T__
+template <> struct from_chars_caller<std::float64_t> {
+  template <typename UC>
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 static from_chars_result_t<UC>
+  call(UC const *first, UC const *last, std::float64_t &value,
+       parse_options_t<UC> options) noexcept {
+    // if std::float64_t is defined, and we are in C++23 mode; macro set for
+    // float64; set value as double due to equivalence between double and
+    // float64_t
+    double val = 0.0;
+    auto ret = from_chars_advanced(first, last, val, options);
+    value = val;
+    return ret;
   }
-  return n;
+};
+#endif
+
+template <typename T, typename UC, typename>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+from_chars(UC const *first, UC const *last, T &value,
+           chars_format fmt /*= chars_format::general*/) noexcept {
+  return from_chars_caller<T>::call(first, last, value,
+                                    parse_options_t<UC>(fmt));
 }

-// computes h * 2^-shift
-void decimal_left_shift(decimal &h, uint32_t shift) {
-  if (h.num_digits == 0) {
-    return;
-  }
-  uint32_t num_new_digits = number_of_digits_decimal_left_shift(h, shift);
-  int32_t read_index = int32_t(h.num_digits - 1);
-  uint32_t write_index = h.num_digits - 1 + num_new_digits;
-  uint64_t n = 0;
-
-  while (read_index >= 0) {
-    n += uint64_t(h.digits[read_index]) << shift;
-    uint64_t quotient = n / 10;
-    uint64_t remainder = n - (10 * quotient);
-    if (write_index < max_digits) {
-      h.digits[write_index] = uint8_t(remainder);
-    } else if (remainder > 0) {
-      h.truncated = true;
-    }
-    n = quotient;
-    write_index--;
-    read_index--;
-  }
-  while (n > 0) {
-    uint64_t quotient = n / 10;
-    uint64_t remainder = n - (10 * quotient);
-    if (write_index < max_digits) {
-      h.digits[write_index] = uint8_t(remainder);
-    } else if (remainder > 0) {
-      h.truncated = true;
-    }
-    n = quotient;
-    write_index--;
-  }
-  h.num_digits += num_new_digits;
-  if (h.num_digits > max_digits) {
-    h.num_digits = max_digits;
-  }
-  h.decimal_point += int32_t(num_new_digits);
-  trim(h);
-}
-
-// computes h * 2^shift
-void decimal_right_shift(decimal &h, uint32_t shift) {
-  uint32_t read_index = 0;
-  uint32_t write_index = 0;
-
-  uint64_t n = 0;
-
-  while ((n >> shift) == 0) {
-    if (read_index < h.num_digits) {
-      n = (10 * n) + h.digits[read_index++];
-    } else if (n == 0) {
-      return;
+template <typename T>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 bool
+clinger_fast_path_impl(uint64_t mantissa, int64_t exponent, bool is_negative,
+                       T &value) noexcept {
+  // The implementation of the Clinger's fast path is convoluted because
+  // we want round-to-nearest in all cases, irrespective of the rounding mode
+  // selected on the thread.
+  // We proceed optimistically, assuming that detail::rounds_to_nearest()
+  // returns true.
+  if (binary_format<T>::min_exponent_fast_path() <= exponent &&
+      exponent <= binary_format<T>::max_exponent_fast_path() &&
+      mantissa <= binary_format<T>::max_mantissa_fast_path()) {
+    // The mantissa bound above is a necessary condition for BOTH branches
+    // below: the rounding-mode-dependent branch checks the tighter
+    // max_mantissa_fast_path(exponent) <= max_mantissa_fast_path(). Testing
+    // it before detail::rounds_to_nearest() spares long-mantissa inputs
+    // (which can never take the fast path) the volatile-float probe.
+    //
+    // Unfortunately, the conventional Clinger's fast path is only possible
+    // when the system rounds to the nearest float.
+    //
+    // We expect the next branch to almost always be selected.
+    // We could check it first (before the previous branch), but
+    // there might be performance advantages at having the check
+    // be last.
+    if (!cpp20_and_in_constexpr() && detail::rounds_to_nearest()) {
+      // We have that fegetround() == FE_TONEAREST.
+      // Next is Clinger's fast path.
+      value = T(mantissa);
+      if (exponent < 0) {
+        value = value / binary_format<T>::exact_power_of_ten(-exponent);
+      } else {
+        value = value * binary_format<T>::exact_power_of_ten(exponent);
+      }
+      if (is_negative) {
+        value = -value;
+      }
+      return true;
     } else {
-      while ((n >> shift) == 0) {
-        n = 10 * n;
-        read_index++;
+      // We do not have that fegetround() == FE_TONEAREST.
+      // Next is a modified Clinger's fast path, inspired by Jakub Jelinek's
+      // proposal
+      if (exponent >= 0 &&
+          mantissa <= binary_format<T>::max_mantissa_fast_path(exponent)) {
+#if defined(__clang__) || defined(SIMDJSON_FASTFLOAT_32BIT)
+        // Clang may map 0 to -0.0 when fegetround() == FE_DOWNWARD
+        if (mantissa == 0) {
+          value = is_negative ? T(-0.) : T(0.);
+          return true;
+        }
+#endif
+        value = T(mantissa) * binary_format<T>::exact_power_of_ten(exponent);
+        if (is_negative) {
+          value = -value;
+        }
+        return true;
       }
-      break;
     }
   }
-  h.decimal_point -= int32_t(read_index - 1);
-  if (h.decimal_point < -decimal_point_range) { // it is zero
-    h.num_digits = 0;
-    h.decimal_point = 0;
-    h.negative = false;
-    h.truncated = false;
-    return;
+  return false;
+}
+
+/**
+ * This function overload takes parsed_number_string_t structure that is created
+ * and populated either by from_chars_advanced function taking chars range and
+ * parsing options or other parsing custom function implemented by user.
+ */
+template <typename T, typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+from_chars_advanced(parsed_number_string_t<UC> &pns, T &value) noexcept {
+  static_assert(is_supported_float_type<T>::value,
+                "only some floating-point types are supported");
+  static_assert(is_supported_char_type<UC>::value,
+                "only char, wchar_t, char16_t and char32_t are supported");
+
+  from_chars_result_t<UC> answer;
+
+  answer.ec = std::errc(); // be optimistic
+  answer.ptr = pns.lastmatch;
+
+  if (!pns.too_many_digits &&
+      clinger_fast_path_impl(pns.mantissa, pns.exponent, pns.negative, value))
+    return answer;
+
+  adjusted_mantissa am =
+      compute_float<binary_format<T>>(pns.exponent, pns.mantissa);
+  if (pns.too_many_digits && am.power2 >= 0) {
+    if (am != compute_float<binary_format<T>>(pns.exponent, pns.mantissa + 1)) {
+      am = compute_error<binary_format<T>>(pns.exponent, pns.mantissa);
+    }
   }
-  uint64_t mask = (uint64_t(1) << shift) - 1;
-  while (read_index < h.num_digits) {
-    uint8_t new_digit = uint8_t(n >> shift);
-    n = (10 * (n & mask)) + h.digits[read_index++];
-    h.digits[write_index++] = new_digit;
+  // If we called compute_float<binary_format<T>>(pns.exponent, pns.mantissa)
+  // and we have an invalid power (am.power2 < 0), then we need to go the long
+  // way around again. This is very uncommon.
+  if (am.power2 < 0) {
+    am = digit_comp<T>(pns, am);
   }
-  while (n > 0) {
-    uint8_t new_digit = uint8_t(n >> shift);
-    n = 10 * (n & mask);
-    if (write_index < max_digits) {
-      h.digits[write_index++] = new_digit;
-    } else if (new_digit > 0) {
-      h.truncated = true;
-    }
+  to_float(pns.negative, am, value);
+  // Test for over/underflow.
+  if ((pns.mantissa != 0 && am.mantissa == 0 && am.power2 == 0) ||
+      am.power2 == binary_format<T>::infinite_power()) {
+    answer.ec = std::errc::result_out_of_range;
   }
-  h.num_digits = write_index;
-  trim(h);
+  return answer;
 }

-template <typename binary> adjusted_mantissa compute_float(decimal &d) {
-  adjusted_mantissa answer;
-  if (d.num_digits == 0) {
-    // should be zero
-    answer.power2 = 0;
-    answer.mantissa = 0;
-    return answer;
-  }
-  // At this point, going further, we can assume that d.num_digits > 0.
-  // We want to guard against excessive decimal point values because
-  // they can result in long running times. Indeed, we do
-  // shifts by at most 60 bits. We have that log(10**400)/log(2**60) ~= 22
-  // which is fine, but log(10**299995)/log(2**60) ~= 16609 which is not
-  // fine (runs for a long time).
-  //
-  if(d.decimal_point < -324) {
-    // We have something smaller than 1e-324 which is always zero
-    // in binary64 and binary32.
-    // It should be zero.
-    answer.power2 = 0;
-    answer.mantissa = 0;
-    return answer;
-  } else if(d.decimal_point >= 310) {
-    // We have something at least as large as 0.1e310 which is
-    // always infinite.
-    answer.power2 = binary::infinite_power();
-    answer.mantissa = 0;
+// Slow path: re-parse materializing the integer/fraction spans the hot no-span
+// parse skipped, then run the full algorithm. The two callers reach it only
+// through a simdjson_fastfloat_unlikely branch, so the optimizer keeps this re-parse off
+// the hot path on its own (no function-level noinline needed).
+// from_chars_advanced already handles both the too_many_digits disambiguation
+// and the am.power2<0 digit_comp recompute, so both slow branches collapse to
+// one helper call.
+template <typename T, typename UC>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+parse_number_slow_path(UC const *first, UC const *last, T &value,
+                       parse_options_t<UC> options, bool bjf) noexcept {
+  parsed_number_string_t<UC> pns =
+      bjf ? parse_number_string<true, UC>(first, last, options, true)
+          : parse_number_string<false, UC>(first, last, options, true);
+  return from_chars_advanced(pns, value);
+}
+
+template <typename T, typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+from_chars_float_advanced(UC const *first, UC const *last, T &value,
+                          parse_options_t<UC> options) noexcept {
+
+  static_assert(is_supported_float_type<T>::value,
+                "only some floating-point types are supported");
+  static_assert(is_supported_char_type<UC>::value,
+                "only char, wchar_t, char16_t and char32_t are supported");
+
+  chars_format const fmt = detail::adjust_for_feature_macros(options.format);
+
+  from_chars_result_t<UC> answer;
+  if (uint64_t(fmt & chars_format::skip_white_space)) {
+    while ((first != last) && simdjson_fast_float::is_space(*first)) {
+      first++;
+    }
+  }
+  if (first == last) {
+    answer.ec = std::errc::invalid_argument;
+    answer.ptr = first;
     return answer;
   }
-
-  static const uint32_t max_shift = 60;
-  static const uint32_t num_powers = 19;
-  static const uint8_t powers[19] = {
-      0,  3,  6,  9,  13, 16, 19, 23, 26, 29, //
-      33, 36, 39, 43, 46, 49, 53, 56, 59,     //
-  };
-  int32_t exp2 = 0;
-  while (d.decimal_point > 0) {
-    uint32_t n = uint32_t(d.decimal_point);
-    uint32_t shift = (n < num_powers) ? powers[n] : max_shift;
-    decimal_right_shift(d, shift);
-    if (d.decimal_point < -decimal_point_range) {
-      // should be zero
-      answer.power2 = 0;
-      answer.mantissa = 0;
+  bool const bjf = uint64_t(fmt & detail::basic_json_fmt) != 0;
+
+  // Fast path: parse WITHOUT materializing the integer/fraction spans (read
+  // only by the rare slow paths). Skipping their stores keeps the fat
+  // parsed_number_string_t off the hot path. store_spans is a runtime argument,
+  // so this reuses the single parse_number_string instantiation.
+  parsed_number_string_t<UC> pns =
+      bjf ? parse_number_string<true, UC>(first, last, options, false)
+          : parse_number_string<false, UC>(first, last, options, false);
+  if (!pns.valid) {
+    if (uint64_t(fmt & chars_format::no_infnan)) {
+      answer.ec = std::errc::invalid_argument;
+      answer.ptr = first;
       return answer;
-    }
-    exp2 += int32_t(shift);
-  }
-  // We shift left toward [1/2 ... 1].
-  while (d.decimal_point <= 0) {
-    uint32_t shift;
-    if (d.decimal_point == 0) {
-      if (d.digits[0] >= 5) {
-        break;
-      }
-      shift = (d.digits[0] < 2) ? 2 : 1;
     } else {
-      uint32_t n = uint32_t(-d.decimal_point);
-      shift = (n < num_powers) ? powers[n] : max_shift;
+      return detail::parse_infnan(first, last, value, fmt);
     }
-    decimal_left_shift(d, shift);
-    if (d.decimal_point > decimal_point_range) {
-      // we want to get infinity:
-      answer.power2 = 0xFF;
-      answer.mantissa = 0;
-      return answer;
-    }
-    exp2 -= int32_t(shift);
   }
-  // We are now in the range [1/2 ... 1] but the binary format uses [1 ... 2].
-  exp2--;
-  constexpr int32_t minimum_exponent = binary::minimum_exponent();
-  while ((minimum_exponent + 1) > exp2) {
-    uint32_t n = uint32_t((minimum_exponent + 1) - exp2);
-    if (n > max_shift) {
-      n = max_shift;
-    }
-    decimal_right_shift(d, n);
-    exp2 += int32_t(n);
+
+  // Slow path A (rare): > 19 significant digits. The no-span parse left the
+  // mantissa un-truncated and skipped the span-based recompute; the cold helper
+  // re-parses with spans and runs the full algorithm.
+  //
+// We have to disable -Wc++20-extensions for the [[unlikely]] attribute
+// See comment for @jwakely at
+// https://github.com/fastfloat/simdjson_fast_float/pull/387#discussion_r3366943539
+// This is unfortunate.
+#ifdef __clang__
+#pragma clang diagnostic push
+#if (!defined(__APPLE_CC__) && __clang_major__ >= 10) || (__clang_major__ >= 13)
+#pragma clang diagnostic ignored "-Wc++20-extensions"
+#endif
+#endif
+  if simdjson_fastfloat_unlikely (pns.too_many_digits) {
+    return parse_number_slow_path<T, UC>(first, last, value, options, bjf);
   }
-  if ((exp2 - minimum_exponent) >= binary::infinite_power()) {
-    answer.power2 = binary::infinite_power();
-    answer.mantissa = 0;
+  answer.ec = std::errc(); // be optimistic
+  answer.ptr = pns.lastmatch;
+
+  if (clinger_fast_path_impl(pns.mantissa, pns.exponent, pns.negative, value)) {
     return answer;
   }

-  const int mantissa_size_in_bits = binary::mantissa_explicit_bits() + 1;
-  decimal_left_shift(d, mantissa_size_in_bits);
-
-  uint64_t mantissa = round(d);
-  // It is possible that we have an overflow, in which case we need
-  // to shift back.
-  if (mantissa >= (uint64_t(1) << mantissa_size_in_bits)) {
-    decimal_right_shift(d, 1);
-    exp2 += 1;
-    mantissa = round(d);
-    if ((exp2 - minimum_exponent) >= binary::infinite_power()) {
-      answer.power2 = binary::infinite_power();
-      answer.mantissa = 0;
-      return answer;
-    }
+  adjusted_mantissa am =
+      compute_float<binary_format<T>>(pns.exponent, pns.mantissa);
+  // Slow path B (rare): Eisel-Lemire could not resolve; digit_comp needs the
+  // integer/fraction spans. Route to the cold helper (clinger there is a
+  // dead-effect since it already failed here; the cold re-parse + digit_comp
+  // via from_chars_advanced reproduces this branch).
+  if simdjson_fastfloat_unlikely (am.power2 < 0) {
+    return parse_number_slow_path<T, UC>(first, last, value, options, bjf);
   }
-  answer.power2 = exp2 - binary::minimum_exponent();
-  if (mantissa < (uint64_t(1) << binary::mantissa_explicit_bits())) {
-    answer.power2--;
+#ifdef __clang__
+#pragma clang diagnostic pop
+#endif
+  to_float(pns.negative, am, value);
+  // Test for over/underflow.
+  if ((pns.mantissa != 0 && am.mantissa == 0 && am.power2 == 0) ||
+      am.power2 == binary_format<T>::infinite_power()) {
+    answer.ec = std::errc::result_out_of_range;
   }
-  answer.mantissa =
-      mantissa & ((uint64_t(1) << binary::mantissa_explicit_bits()) - 1);
   return answer;
 }

-template <typename binary>
-adjusted_mantissa parse_long_mantissa(const char *first) {
-  decimal d = parse_decimal(first);
-  return compute_float<binary>(d);
+template <typename T, typename UC, typename>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+from_chars(UC const *first, UC const *last, T &value, int base) noexcept {
+
+  static_assert(is_supported_integer_type<T>::value,
+                "only integer types are supported");
+  static_assert(is_supported_char_type<UC>::value,
+                "only char, wchar_t, char16_t and char32_t are supported");
+
+  parse_options_t<UC> options;
+  options.base = base;
+  return from_chars_advanced(first, last, value, options);
 }

-template <typename binary>
-adjusted_mantissa parse_long_mantissa(const char *first, const char *end) {
-  decimal d = parse_decimal(first, end);
-  return compute_float<binary>(d);
+template <typename T>
+SIMDJSON_FASTFLOAT_CONSTEXPR20
+    typename std::enable_if<is_supported_float_type<T>::value, T>::type
+    integer_times_pow10(uint64_t mantissa, int decimal_exponent) noexcept {
+  T value;
+  if (clinger_fast_path_impl(mantissa, decimal_exponent, false, value))
+    return value;
+
+  adjusted_mantissa am =
+      compute_float<binary_format<T>>(decimal_exponent, mantissa);
+  to_float(false, am, value);
+  return value;
 }

-double from_chars(const char *first) noexcept {
-  bool negative = first[0] == '-';
-  if (negative) {
-    first++;
-  }
-  adjusted_mantissa am = parse_long_mantissa<binary_format<double>>(first);
-  uint64_t word = am.mantissa;
-  word |= uint64_t(am.power2)
-          << binary_format<double>::mantissa_explicit_bits();
-  word = negative ? word | (uint64_t(1) << binary_format<double>::sign_index())
-                  : word;
-  double value;
-  std::memcpy(&value, &word, sizeof(double));
+template <typename T>
+SIMDJSON_FASTFLOAT_CONSTEXPR20
+    typename std::enable_if<is_supported_float_type<T>::value, T>::type
+    integer_times_pow10(int64_t mantissa, int decimal_exponent) noexcept {
+  const bool is_negative = mantissa < 0;
+  const uint64_t m = static_cast<uint64_t>(is_negative ? -mantissa : mantissa);
+
+  T value;
+  if (clinger_fast_path_impl(m, decimal_exponent, is_negative, value))
+    return value;
+
+  adjusted_mantissa am = compute_float<binary_format<T>>(decimal_exponent, m);
+  to_float(is_negative, am, value);
   return value;
 }

+SIMDJSON_FASTFLOAT_CONSTEXPR20 inline double
+integer_times_pow10(uint64_t mantissa, int decimal_exponent) noexcept {
+  return integer_times_pow10<double>(mantissa, decimal_exponent);
+}

-double from_chars(const char *first, const char *end) noexcept {
-  bool negative = first[0] == '-';
-  if (negative) {
-    first++;
+SIMDJSON_FASTFLOAT_CONSTEXPR20 inline double
+integer_times_pow10(int64_t mantissa, int decimal_exponent) noexcept {
+  return integer_times_pow10<double>(mantissa, decimal_exponent);
+}
+
+// the following overloads are here to avoid surprising ambiguity for int,
+// unsigned, etc.
+template <typename T, typename Int>
+SIMDJSON_FASTFLOAT_CONSTEXPR20
+    typename std::enable_if<is_supported_float_type<T>::value &&
+                                std::is_integral<Int>::value &&
+                                !std::is_signed<Int>::value,
+                            T>::type
+    integer_times_pow10(Int mantissa, int decimal_exponent) noexcept {
+  return integer_times_pow10<T>(static_cast<uint64_t>(mantissa),
+                                decimal_exponent);
+}
+
+template <typename T, typename Int>
+SIMDJSON_FASTFLOAT_CONSTEXPR20
+    typename std::enable_if<is_supported_float_type<T>::value &&
+                                std::is_integral<Int>::value &&
+                                std::is_signed<Int>::value,
+                            T>::type
+    integer_times_pow10(Int mantissa, int decimal_exponent) noexcept {
+  return integer_times_pow10<T>(static_cast<int64_t>(mantissa),
+                                decimal_exponent);
+}
+
+template <typename Int>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 typename std::enable_if<
+    std::is_integral<Int>::value && !std::is_signed<Int>::value, double>::type
+integer_times_pow10(Int mantissa, int decimal_exponent) noexcept {
+  return integer_times_pow10(static_cast<uint64_t>(mantissa), decimal_exponent);
+}
+
+template <typename Int>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 typename std::enable_if<
+    std::is_integral<Int>::value && std::is_signed<Int>::value, double>::type
+integer_times_pow10(Int mantissa, int decimal_exponent) noexcept {
+  return integer_times_pow10(static_cast<int64_t>(mantissa), decimal_exponent);
+}
+
+template <typename T, typename UC>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+from_chars_int_advanced(UC const *first, UC const *last, T &value,
+                        parse_options_t<UC> options) noexcept {
+
+  static_assert(is_supported_integer_type<T>::value,
+                "only integer types are supported");
+  static_assert(is_supported_char_type<UC>::value,
+                "only char, wchar_t, char16_t and char32_t are supported");
+
+  chars_format const fmt = detail::adjust_for_feature_macros(options.format);
+  int const base = options.base;
+
+  from_chars_result_t<UC> answer;
+  if (uint64_t(fmt & chars_format::skip_white_space)) {
+    while ((first != last) && simdjson_fast_float::is_space(*first)) {
+      first++;
+    }
+  }
+  if (first == last || base < 2 || base > 36) {
+    answer.ec = std::errc::invalid_argument;
+    answer.ptr = first;
+    return answer;
+  }
+
+  return parse_int_string(first, last, value, options);
+}
+
+template <size_t TypeIx> struct from_chars_advanced_caller {
+  static_assert(TypeIx > 0, "unsupported type");
+};
+
+template <> struct from_chars_advanced_caller<1> {
+  template <typename T, typename UC>
+  simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 static from_chars_result_t<UC>
+  call(UC const *first, UC const *last, T &value,
+       parse_options_t<UC> options) noexcept {
+    return from_chars_float_advanced(first, last, value, options);
+  }
+};
+
+template <> struct from_chars_advanced_caller<2> {
+  template <typename T, typename UC>
+  simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 static from_chars_result_t<UC>
+  call(UC const *first, UC const *last, T &value,
+       parse_options_t<UC> options) noexcept {
+    return from_chars_int_advanced(first, last, value, options);
+  }
+};
+
+template <typename T, typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+from_chars_advanced(UC const *first, UC const *last, T &value,
+                    parse_options_t<UC> options) noexcept {
+  return from_chars_advanced_caller<
+      size_t(is_supported_float_type<T>::value) +
+      2 * size_t(is_supported_integer_type<T>::value)>::call(first, last, value,
+                                                             options);
+}
+
+} // namespace simdjson_fast_float
+
+#endif
+
+/* end file simdjson/internal/fast_float.h */
+
+#include <cstdint>
+#include <cstring>
+#include <limits>
+
+namespace simdjson {
+namespace internal {
+
+/**
+ * These functions handle floating-point parsing when the fast path in
+ * numberparsing.h gives up: more than 19 digits in the decimal mantissa, an
+ * exponent outside the range the power-of-five table covers, or one of the rare
+ * inputs where the truncated Eisel-Lemire product is not accurate enough to
+ * round. That should only be seen in adversarial scenarios; we do not expect
+ * production systems to even produce such floating-point numbers.
+ *
+ * The work is handed to fast_float (vendored in
+ * include/simdjson/internal/fast_float.h), which settles the rounding by
+ * comparing a bigint against a scaled power of five. It is correctly rounded,
+ * and quick enough that an adversarial document is no longer worth worrying
+ * about.
+ **/
+
+namespace {
+
+// fast_float wants the end of the number, and the callers only promise that a
+// number is followed by a character which cannot be part of one -- the input
+// has already been validated against the JSON grammar, and the buffer is padded,
+// so such a character is always there to be found. Locating it costs a pass over
+// digits we are about to parse anyway, and in exchange fast_float can bound its
+// inner loops instead of re-checking a far-away end pointer.
+const char *find_end_of_number(const char *first) noexcept {
+  const char *p = first;
+  while ((*p >= '0' && *p <= '9') || *p == '-' || *p == '+' || *p == '.' ||
+         *p == 'e' || *p == 'E') {
+    p++;
   }
-  adjusted_mantissa am = parse_long_mantissa<binary_format<double>>(first, end);
-  uint64_t word = am.mantissa;
-  word |= uint64_t(am.power2)
-          << binary_format<double>::mantissa_explicit_bits();
-  word = negative ? word | (uint64_t(1) << binary_format<double>::sign_index())
-                  : word;
-  double value;
-  std::memcpy(&value, &word, sizeof(double));
+  return p;
+}
+
+// The input is JSON, so parse it under the JSON grammar: no hexadecimal, no
+// leading plus, and no infinity or NaN spellings. Those are handled (or
+// rejected) before we ever get here.
+constexpr simdjson_fast_float::parse_options json_options{
+    simdjson_fast_float::chars_format::json};
+
+// fast_float reports result_out_of_range for a value at either edge of the
+// format, writing +/-0 when it underflows and +/-infinity when it overflows.
+// Both are exactly what the callers of these functions expect to receive: they
+// accept a zero and treat an infinity as an error. A malformed number cannot
+// happen on validated input, but if it somehow did, returning zero matches what
+// the previous implementation did with digits it could not use.
+template <typename T> T parse_with_fast_float(const char *first, const char *end) noexcept {
+  T value{};
+  auto answer =
+      simdjson_fast_float::from_chars_advanced(first, end, value, json_options);
+  if (answer.ec == std::errc::invalid_argument) { return T(0); }
   return value;
 }

+} // namespace
+
+double from_chars(const char *first) noexcept {
+  return parse_with_fast_float<double>(first, find_end_of_number(first));
+}
+
+double from_chars(const char *first, const char *end) noexcept {
+  return parse_with_fast_float<double>(first, end);
+}
+
+float from_chars_float(const char *first) noexcept {
+  return parse_with_fast_float<float>(first, find_end_of_number(first));
+}
+
 } // internal
 } // simdjson

@@ -5109,7 +10245,8 @@ namespace internal {
     { SCALAR_DOCUMENT_AS_VALUE, "SCALAR_DOCUMENT_AS_VALUE: A JSON document made of a scalar (number, Boolean, null or string) is treated as a value. Use get_bool(), get_double(), etc. on the document instead. "},
     { OUT_OF_BOUNDS, "OUT_OF_BOUNDS: Attempt to access location outside of document."},
     { TRAILING_CONTENT, "TRAILING_CONTENT: Unexpected trailing content in the JSON input."},
-    { OUT_OF_CAPACITY, "OUT_OF_CAPACITY: The capacity was exceeded, we cannot allocate enough memory."}
+    { OUT_OF_CAPACITY, "OUT_OF_CAPACITY: The capacity was exceeded, we cannot allocate enough memory."},
+    { UNKNOWN_FIELD, "UNKNOWN_FIELD: The JSON object has a field that does not map to any member of the target type (deny_unknown_fields)."}
   }; // error_messages[]

 } // namespace internal
@@ -7115,7 +12252,12 @@ class document;
 * 3) The stream_final mode allows us to truncate final
 * unterminated strings. It is useful in conjunction with streaming_partial.
 */
-enum class stage1_mode { regular, streaming_partial, streaming_final};
+enum class stage1_mode {
+  regular,
+  streaming_partial, streaming_final,
+  json_sequence_partial, json_sequence_final,
+  comma_delimited_partial, comma_delimited_final
+};

 /**
  * Returns true if mode == streaming_partial or mode == streaming_final
@@ -7127,7 +12269,6 @@ inline bool is_streaming(stage1_mode mode) {
   // return (mode == stage1_mode::streaming_partial || mode == stage1_mode::streaming_final);
 }

-
 namespace internal {


@@ -7312,6 +12453,16 @@ public:
   /** Whether to store big integers as strings instead of returning BIGINT_ERROR */
   bool _number_as_string{false};

+  /**
+   * Whether the input buffer passed to parse() is *not* padded to len +
+   * SIMDJSON_PADDING bytes. When true, stage 2 string parsing avoids reading
+   * past buf+len (it finishes the final, near-the-end bytes from a small padded
+   * scratch buffer). This is set only by the no-padding DOM parse entry points
+   * (dom::parser::parse_unpadded); the default padded fast path leaves it false
+   * and is unaffected.
+   */
+  bool _unpadded{false};
+
 protected:

   // Declaring these so that subclasses can use them to implement their constructors.
@@ -7655,6 +12806,8 @@ enum instruction_set {
   LASX = 0x40000,
   //RVV = 0x80000,
   RVV_VLS = 0x100000,
+  SVE = 0x200000,
+  SVE2 = 0x400000,
 };

 } // namespace internal
@@ -7966,12 +13119,24 @@ POSSIBILITY OF SUCH DAMAGE.
 #include <cstdlib>
 #if defined(_MSC_VER)
 #include <intrin.h>
-#elif defined(HAVE_GCC_GET_CPUID) && defined(USE_GCC_GET_CPUID)
+#elif (defined(HAVE_GCC_GET_CPUID) && defined(USE_GCC_GET_CPUID)) || defined(__FILC__)
 #include <cpuid.h>
 #endif
 #if defined(__loongarch__) && defined(__linux__)
   #include <sys/auxv.h>
 #endif
+#if defined(__aarch64__) && defined(__linux__)
+  #include <sys/auxv.h>
+#endif
+#if (defined(__aarch64__) || defined(_M_ARM64) || defined(_M_ARM64EC)) && defined(_WIN32) && !defined(_WINDOWS_)
+// We avoid including <windows.h> (macro pollution); this matches the
+// declaration in the Windows SDK (BOOL WINAPI IsProcessorFeaturePresent(DWORD)).
+extern "C" __declspec(dllimport) int __stdcall IsProcessorFeaturePresent(unsigned long ProcessorFeature);
+#endif
+
+#ifdef __FILC__
+#include <stdfil.h>
+#endif

 namespace simdjson {
 namespace internal {
@@ -7984,8 +13149,60 @@ static inline uint32_t detect_supported_architectures() {

 #elif defined(__aarch64__) || defined(_M_ARM64) || defined(_M_ARM64EC)

+#if defined(__linux__)
+// The kernel advertises SVE in AT_HWCAP and SVE2 in AT_HWCAP2. Older
+// headers may not define these constants, so we provide the kernel's values
+// (we deliberately do not include <asm/hwcap.h>, which is not available on
+// all toolchains, e.g., musl without linux-headers).
+#ifndef AT_HWCAP2
+#define AT_HWCAP2 26
+#endif
+#ifndef HWCAP_SVE
+#define HWCAP_SVE (1 << 22)
+#endif
+#ifndef HWCAP2_SVE2
+#define HWCAP2_SVE2 (1 << 1)
+#endif
+#endif // __linux__
+
+#if defined(_WIN32)
+// Only recent Windows SDKs define these processor features.
+#ifndef PF_ARM_SVE_INSTRUCTIONS_AVAILABLE
+#define PF_ARM_SVE_INSTRUCTIONS_AVAILABLE 46
+#endif
+#ifndef PF_ARM_SVE2_INSTRUCTIONS_AVAILABLE
+#define PF_ARM_SVE2_INSTRUCTIONS_AVAILABLE 47
+#endif
+#endif // _WIN32
+
 static inline uint32_t detect_supported_architectures() {
-  return instruction_set::NEON;
+  // NEON is mandatory on AArch64.
+  uint32_t host_isa = instruction_set::NEON;
+#if defined(__linux__)
+  unsigned long hwcap = getauxval(AT_HWCAP);
+  unsigned long hwcap2 = getauxval(AT_HWCAP2);
+  if (hwcap & HWCAP_SVE) {
+    host_isa |= instruction_set::SVE;
+    // We only claim SVE2 when SVE is also present. Before Linux 6.14, the
+    // kernel set HWCAP2_SVE2 on processors implementing SME(2) but not SVE,
+    // because SVE2 instructions are available in streaming mode. Our SVE2
+    // code runs in non-streaming mode and needs actual SVE.
+    if (hwcap2 & HWCAP2_SVE2) {
+      host_isa |= instruction_set::SVE2;
+    }
+  }
+#elif defined(_WIN32)
+  if (IsProcessorFeaturePresent(PF_ARM_SVE_INSTRUCTIONS_AVAILABLE)) {
+    host_isa |= instruction_set::SVE;
+    // As on Linux, require SVE before claiming SVE2.
+    if (IsProcessorFeaturePresent(PF_ARM_SVE2_INSTRUCTIONS_AVAILABLE)) {
+      host_isa |= instruction_set::SVE2;
+    }
+  }
+#endif
+  // On other systems (e.g., macOS, where Apple Silicon has no SVE), we only
+  // report NEON.
+  return host_isa;
 }

 #elif defined(__x86_64__) || defined(_M_AMD64) // x64
@@ -8023,7 +13240,7 @@ static inline void cpuid(uint32_t *eax, uint32_t *ebx, uint32_t *ecx,
   *ebx = cpu_info[1];
   *ecx = cpu_info[2];
   *edx = cpu_info[3];
-#elif defined(HAVE_GCC_GET_CPUID) && defined(USE_GCC_GET_CPUID)
+#elif (defined(HAVE_GCC_GET_CPUID) && defined(USE_GCC_GET_CPUID)) || defined(__FILC__)
   uint32_t level = *eax;
   __get_cpuid(level, eax, ebx, ecx, edx);
 #else
@@ -8040,6 +13257,8 @@ static inline void cpuid(uint32_t *eax, uint32_t *ebx, uint32_t *ecx,
 static inline uint64_t xgetbv() {
 #if defined(_MSC_VER)
   return _xgetbv(0);
+#elif defined(__FILC__)
+  return zxgetbv();
 #else
   uint32_t xcr0_lo, xcr0_hi;
   asm volatile("xgetbv\n\t" : "=a" (xcr0_lo), "=d" (xcr0_hi) : "c" (0));
@@ -8601,7 +13820,7 @@ public:
   simdjson_inline implementation() : simdjson::implementation(
       "rvv_vls",
       "RISC-V V extension",
-      0
+      internal::instruction_set::RVV_VLS
   ) {}
   simdjson_warn_unused error_code create_dom_parser_implementation(
     size_t capacity,
@@ -8976,7 +14195,14 @@ simdjson_inline int leading_zeroes(uint64_t input_num) {

 /* result might be undefined when input_num is zero */
 simdjson_inline int count_ones(uint64_t input_num) {
+#if SIMDJSON_REGULAR_VISUAL_STUDIO
    return vaddv_u8(vcnt_u8(vcreate_u8(input_num)));
+#else
+   // if the system supports SVE or CSSC, __builtin_popcountll
+   // might be compiled to fewer single instructions. For CSSC,
+   // __builtin_popcountll is compiled to a single instruction.
+   return __builtin_popcountll(input_num);
+#endif// SIMDJSON_REGULAR_VISUAL_STUDIO
 }


@@ -9013,15 +14239,6 @@ simdjson_inline uint64_t zero_leading_bit(uint64_t rev_bits, int leading_zeroes)

 #endif

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2, uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
-  *result = value1 + value2;
-  return *result < value1;
-#else
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-#endif
-}

 } // unnamed namespace
 } // namespace arm64
@@ -9285,6 +14502,7 @@ namespace {
       return vget_lane_u64(
           vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(*this), 4)), 0);
     }
+    // Integer umaxv, not a float compare: flush-to-zero would treat a denormal as zero.
     simdjson_inline bool any() const { return vmaxvq_u32(vreinterpretq_u32_u8(*this)) != 0; }
   };

@@ -9933,6 +15151,9 @@ namespace atomparsing {
 // to the compile-time constant 1936482662.
 simdjson_inline uint32_t string_to_uint32(const char* str) { uint32_t val; std::memcpy(&val, str, sizeof(uint32_t)); return val; }

+// Acts on the same principle as string_to_uint32, but on an 8-byte block of memory
+simdjson_inline uint64_t string_to_uint64(const char* str) { uint64_t val; std::memcpy(&val, str, sizeof(uint64_t)); return val; }
+

 // Again in str4ncmp we use a memcpy to avoid undefined behavior. The memcpy may appear expensive.
 // Yet all decent optimizing compilers will compile memcpy to a single instruction, just about.
@@ -9944,6 +15165,28 @@ simdjson_inline uint32_t str4ncmp(const uint8_t *src, const char* atom) {
   return srcval ^ string_to_uint32(atom);
 }

+// Checks that the first 8 characters of the input string match the given atom in a case-insensitive manner.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint64_t str8ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+  uint64_t srcval; // we want to avoid unaligned 32-bit loads (undefined in C/C++)
+  static_assert(sizeof(uint64_t) <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be larger than 8 bytes");
+  std::memcpy(&srcval, src, sizeof(uint64_t));
+
+  return (srcval | 0x2020202020202020ull) ^ string_to_uint64(atom);
+}
+
+// Checks that the first 3 characters of 'src' match 'atom' in a case-insensitive way.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint32_t str3ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+  return ((src[0] | 0x20) ^ atom[0]) //
+       | ((src[1] | 0x20) ^ atom[1]) //
+       | ((src[2] | 0x20) ^ atom[2]);
+}
+
 simdjson_warn_unused
 simdjson_inline bool is_valid_true_atom(const uint8_t *src) {
   return (str4ncmp(src, "true") | jsoncharutils::is_not_structural_or_whitespace(src[4])) == 0;
@@ -9980,6 +15223,71 @@ simdjson_inline bool is_valid_null_atom(const uint8_t *src, size_t len) {
   else { return false; }
 }

+#if SIMDJSON_ENABLE_NAN_INF
+// "nan" is 3 bytes; we check characters and then verify the next
+// character is structural or whitespace. We accept both "nan" and "NaN".
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src) {
+  return (str3ncmp_case_insensitive(src, "nan")
+        | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+}
+
+// checks that the next four characters of a string are 'nan"', where the 'nan'
+// is checked in a case-insensitive way.
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_in_string(const uint8_t *src) {
+  return (str3ncmp_case_insensitive(src, "nan") | (src[3] ^ '"')) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src, size_t len) {
+  if (len > 3) { return is_valid_nan_atom(src); }
+  if (len == 3) { return str3ncmp_case_insensitive(src, "nan") == 0; }
+  return false;
+}
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src) {
+  bool is_short_inf = (str3ncmp_case_insensitive(src, "inf")
+                     | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+  if(is_short_inf) return true;
+
+  // Check for 'infinity' (any capitalization)
+  return (str8ncmp_case_insensitive(src, "infinity") | jsoncharutils::is_not_structural_or_whitespace(src[8])) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_in_string(const uint8_t *src) {
+  bool is_short_inf = (str3ncmp_case_insensitive(src, "inf") | (src[3] ^ '"')) == 0;
+  if(is_short_inf) return true;
+
+  return (str8ncmp_case_insensitive(src, "infinity") | (src[8] ^ '"')) == 0;
+}
+
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src, size_t len) {
+  if (len > 8) { return is_valid_inf_atom(src); }
+  if (len == 8 && str8ncmp_case_insensitive(src, "infinity") == 0) {
+    return true;
+  }
+  if (len > 3) {
+    return (str3ncmp_case_insensitive(src, "inf")
+          | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+  }
+  if (len == 3) { return str3ncmp_case_insensitive(src, "inf") == 0; }
+  return false;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
 } // namespace atomparsing
 } // unnamed namespace
 } // namespace arm64
@@ -10070,7 +15378,7 @@ inline simdjson_warn_unused error_code dom_parser_implementation::set_capacity(s
 }

 inline simdjson_warn_unused error_code dom_parser_implementation::set_max_depth(size_t max_depth) noexcept {
-  if(max_depth == 0 || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
+  if(max_depth == 0 || max_depth > SIMDJSON_MAX_DEPTH || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
   // Stage 2 stacks
   open_containers.reset(new (std::nothrow) open_container[max_depth]);
   is_array.reset(new (std::nothrow) bool[max_depth]);
@@ -10249,6 +15557,7 @@ protected:
 /* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_NUMBERPARSING_H */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/base.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/jsoncharutils.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/atomparsing.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/internal/numberparsing_tables.h" */
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */

@@ -10288,6 +15597,18 @@ simdjson_inline double to_double(uint64_t mantissa, uint64_t real_exponent, bool
     return d;
 }

+// Convert a mantissa, an exponent and a sign bit into an ieee32 float (binary32).
+// The real_exponent needs to be in [0, 254] (technically real_exponent = 255 would be acceptable).
+// The mantissa should be in [0,1<<24). The bit at index (1U << 23) will be zeroed.
+simdjson_inline float to_float(uint32_t mantissa, uint32_t real_exponent, bool negative) {
+    float f;
+    mantissa &= ~(uint32_t(1) << 23);
+    mantissa |= real_exponent << 23;
+    mantissa |= ((static_cast<uint32_t>(negative)) << 31);
+    std::memcpy(&f, &mantissa, sizeof(f));
+    return f;
+}
+
 // Attempts to compute i * 10^(power) exactly; and if "negative" is
 // true, negate the result.
 // This function will only work in some cases, when it does not work, success is
@@ -10544,6 +15865,219 @@ simdjson_inline bool compute_float_64(int64_t power, uint64_t i, bool negative,
   return true;
 }

+// Attempts to compute i * 10^(power) as a binary32 (float) value; and if
+// "negative" is true, negate the result.
+//
+// This is the same Eisel-Lemire algorithm as compute_float_64, retargeted at
+// binary32. It is adapted from fast_float
+// (https://github.com/fastfloat/fast_float), where the routine is written once
+// and instantiated for each binary format. We only need 24 bits of mantissa
+// instead of 53, so we ask the truncated product for 24 + 2 = 26 accurate bits
+// (plus one bit that may be lost to the "upperbit" shift), and we shift the
+// product right by (upperbit + 64 - 23 - 3) instead of (upperbit + 64 - 52 - 3).
+//
+// The power-of-five table is shared with the binary64 code: the useful range
+// for binary32, [smallest_power_binary32, largest_power_binary32], is a strict
+// subset of [smallest_power, largest_power].
+//
+// The function returns false when the result would be infinite: simdjson
+// refuses to parse infinite values, so the caller reports an error. Unlike
+// compute_float_64, a false return never means "try harder": the accuracy of
+// the two-limb product is guaranteed by Noble Mushtak and Daniel Lemire,
+// "Fast Number Parsing Without Fallback" (https://arxiv.org/abs/2212.06644).
+// The caller must still fall back to a slow path when the 64-bit mantissa i
+// was truncated (more than 19 significant digits).
+//
+// We assume that power is in the [smallest_power, largest_power] interval: the
+// caller is responsible for this check.
+simdjson_inline bool compute_float_32(int64_t power, uint64_t i, bool negative, float &d) {
+  // Powers of ten that are exactly representable as binary32 values: 10^k is
+  // exact as long as k <= 10 (5^10 = 9765625 < 2**24 while 5^11 > 2**24).
+  static constexpr float power_of_ten_float[] = {
+      1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+      1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+  // The range of powers of ten that a non-zero, finite binary32 value can be
+  // built from. Anything smaller rounds to zero, anything larger is infinite.
+  // These are the smallest_power_of_ten()/largest_power_of_ten() constants that
+  // fast_float uses for binary32.
+  constexpr int smallest_power_binary32 = -65;
+  constexpr int largest_power_binary32 = 38;
+
+  // We start with the fast path described in
+  // Clinger WD. How to read floating point numbers accurately.
+  // ACM SIGPLAN Notices. 1990
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+  // We cannot be certain that x/y is rounded to nearest.
+  if (0 <= power && power <= 10 && i <= 16777215)
+#else
+  if (-10 <= power && power <= 10 && i <= 16777215)
+#endif
+  {
+    // Convert the integer into a float. This is lossless since
+    // 0 <= i <= 2^24 - 1.
+    d = float(i);
+    // Both d and the power of ten are exactly representable as binary32
+    // values, so the product (or quotient) is correctly rounded.
+    if (power < 0) {
+      d = d / power_of_ten_float[-power];
+    } else {
+      d = d * power_of_ten_float[power];
+    }
+    if (negative) {
+      d = -d;
+    }
+    return true;
+  }
+
+  // The fast path has failed, so we fall back on the Eisel-Lemire algorithm.
+  // It needs i > 0 (so that the leading bit of i can be normalized), so we
+  // handle i == 0 separately. We also handle the powers of ten that are so
+  // small (or so large) that the answer is zero (or infinite) whatever the
+  // mantissa is.
+  if (i == 0 || power < smallest_power_binary32) {
+    d = negative ? -0.0f : 0.0f;
+    return true;
+  }
+  if (power > largest_power_binary32) {
+    // We have, for sure, an infinite value.
+    return false;
+  }
+
+  // The exponent is 128 + 63 + power + floor(log(5**power)/log(2)).
+  // The 128 comes from the ieee32 standard (the minimal exponent is -127).
+  // The 63 comes from the fact that we use a 64-bit word.
+  // See compute_float_64 for a discussion of the magical 152170 + 65536.
+  int64_t exponent = (((152170 + 65536) * power) >> 16) + 128 + 63;
+
+  // We want the most significant bit of i to be 1. Shift if needed.
+  int lz = leading_zeroes(i);
+  i <<= lz;
+
+  // We want the most significant 64 bits of the product i * 5**power. It is
+  // safe to index the table because
+  // smallest_power <= smallest_power_binary32 <= power
+  //               <= largest_power_binary32 <= largest_power.
+  const uint32_t index = 2 * uint32_t(power - simdjson::internal::smallest_power);
+#if SIMDJSON_STATIC_REFLECTION
+  simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index]);
+#else
+  simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index]);
+#endif
+
+  // Unless the least significant 38 bits of the high (64-bit) part of the full
+  // product are all 1s, then we know that the most significant 26 bits are
+  // exact and no further work is needed. Having 26 bits is necessary because
+  // we need 24 bits for the mantissa but we have to have one rounding bit and
+  // we can waste a bit if the most significant bit of the product is zero.
+  // (For binary64, the same reasoning gives 55 bits and a 9-bit mask.)
+  if((firstproduct.high & 0x3FFFFFFFFF) == 0x3FFFFFFFFF) {
+    // The truncated multiplication was not accurate enough; use the next 64
+    // bits of the power of five to refine it. See compute_float_64 for a
+    // detailed discussion.
+#if SIMDJSON_STATIC_REFLECTION
+    simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index + 1]);
+#else
+    simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index + 1]);
+#endif
+    firstproduct.low += secondproduct.high;
+    if(secondproduct.high > firstproduct.low) { firstproduct.high++; }
+  }
+  uint64_t lower = firstproduct.low;
+  uint64_t upper = firstproduct.high;
+  // The final mantissa should be 24 bits with a leading 1.
+  // We shift it so that it occupies 25 bits with a leading 1.
+  ///////
+  uint64_t upperbit = upper >> 63;
+  uint64_t mantissa = upper >> (upperbit + 38); // 38 == 64 - 23 - 3
+  lz += int(1 ^ upperbit);
+
+  // Here we have mantissa < (1<<25).
+  int64_t real_exponent = exponent - lz;
+  if (simdjson_unlikely(real_exponent <= 0)) { // we have a subnormal?
+    // Here we have that real_exponent <= 0 so -real_exponent >= 0
+    if(-real_exponent + 1 >= 64) { // if we have more than 64 bits below the minimum exponent, you have a zero for sure.
+      d = negative ? -0.0f : 0.0f;
+      return true;
+    }
+    // next line is safe because -real_exponent + 1 < 64
+    mantissa >>= -real_exponent + 1;
+    // Thankfully, we can't have both "round-to-even" and subnormals because
+    // "round-to-even" only occurs for powers close to 0.
+    mantissa += (mantissa & 1); // round up
+    mantissa >>= 1;
+    // As in compute_float_64, rounding up may take us out of the subnormal
+    // range, so we can only decide after rounding.
+    real_exponent = (mantissa < (uint64_t(1) << 23)) ? 0 : 1;
+    d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+    return true;
+  }
+  // We have to round to even. The "to even" part is only a problem when we are
+  // right in between two floats, which we guard against. The bounds on the
+  // power of ten are those of fast_float's
+  // min_exponent_round_to_even()/max_exponent_round_to_even() for binary32:
+  // when q >= 0, (2m+1) must be divisible by 5^q with 5^q <= 2^25, so q <= 10;
+  // when q < 0, we need 2^24 x 5^{-q} < 2^{64}, so q >= -17.
+  if (simdjson_unlikely((lower <= 1) && (power >= -17) && (power <= 10) && ((mantissa & 3) == 1))) {
+    if((mantissa << (upperbit + 38)) == upper) {
+      mantissa &= ~uint64_t(1);   // flip it so that we do not round up
+    }
+  }
+
+  mantissa += mantissa & 1;
+  mantissa >>= 1;
+
+  // Here we have mantissa < (1<<24), unless there was an overflow
+  if (mantissa >= (uint64_t(1) << 24)) {
+    mantissa = (uint64_t(1) << 23);
+    real_exponent++;
+  }
+  mantissa &= ~(uint64_t(1) << 23);
+  // we have to check that real_exponent is in range, otherwise we bail out
+  if (simdjson_unlikely(real_exponent > 254)) {
+    // We have an infinite value!!! We could actually throw an error here if we could.
+    return false;
+  }
+  d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+  return true;
+}
+
+#if SIMDJSON_ENABLE_NAN_INF
+// Parses a nan or infinity. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, double& d) noexcept {
+  if (atomparsing::is_valid_inf_atom(src)) {
+    double inf = std::numeric_limits<double>::infinity();
+    d = negative ? -inf : inf;
+    return true;
+  }
+
+  if (atomparsing::is_valid_nan_atom(src)) {
+    d = std::numeric_limits<double>::quiet_NaN();
+    return true;
+  }
+
+  return false;
+}
+
+// Parses a nan or infinity as a binary32 value. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, float& d) noexcept {
+  if (atomparsing::is_valid_inf_atom(src)) {
+    float inf = std::numeric_limits<float>::infinity();
+    d = negative ? -inf : inf;
+    return true;
+  }
+
+  if (atomparsing::is_valid_nan_atom(src)) {
+    d = std::numeric_limits<float>::quiet_NaN();
+    return true;
+  }
+
+  return false;
+}
+#endif
+
 // We call a fallback floating-point parser that might be slow. Note
 // it will accept JSON numbers, but the JSON spec. is more restrictive so
 // before you call parse_float_fallback, you need to have validated the input
@@ -10581,6 +16115,16 @@ static bool parse_float_fallback(const uint8_t *ptr, const uint8_t *end_ptr, dou
   return !(*outDouble > (std::numeric_limits<double>::max)() || *outDouble < std::numeric_limits<double>::lowest());
 }

+// Same as parse_float_fallback, but for binary32 (float) values. Going through
+// the binary64 fallback and then rounding to binary32 would be subject to
+// double rounding, so we run the fallback algorithm directly on binary32.
+static bool parse_float_fallback(const uint8_t *ptr, float *outFloat) {
+  *outFloat = simdjson::internal::from_chars_float(reinterpret_cast<const char *>(ptr));
+  // We do not accept infinite values. See the binary64 version above for why we
+  // do not use std::isfinite.
+  return !(*outFloat > (std::numeric_limits<float>::max)() || *outFloat < std::numeric_limits<float>::lowest());
+}
+
 // check quickly whether the next 8 chars are made of digits
 // at a glance, it looks better than Mula's
 // http://0x80.pl/articles/swar-digits-validate.html
@@ -10599,6 +16143,28 @@ simdjson_inline bool is_made_of_eight_digits_fast(const uint8_t *chars) {
           0x3333333333333333);
 }

+// The same idea for four characters. A block of eight only bites when eight
+// digits are left, so a run of 4 to 7 digits used to fall back to one digit at a
+// time; taking four of them at once is what makes the tail of a long fraction
+// cheap. Adapted from fast_float, credit @aqrit.
+simdjson_inline bool is_made_of_four_digits_fast(const uint8_t *chars) {
+  uint32_t val;
+  // Reads up to 3 bytes beyond the digits, which SIMDJSON_PADDING covers.
+  static_assert(3 <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be bigger than 3");
+  std::memcpy(&val, chars, 4);
+  return !(((val + 0x46464646) | (val - 0x30303030)) & 0x80808080);
+}
+
+// Only call this when is_made_of_four_digits_fast() says the four characters are
+// digits. Little-endian: guarded by SIMDJSON_SWAR_NUMBER_PARSING.
+simdjson_inline uint32_t parse_four_digits_unrolled(const uint8_t *chars) {
+  uint32_t val;
+  std::memcpy(&val, chars, 4);
+  val -= 0x30303030;
+  val = (val * 10) + (val >> 8);
+  return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
 template<typename I>
 SIMDJSON_NO_SANITIZE_UNDEFINED // We deliberately allow overflow here and check later
 simdjson_inline bool parse_digit(const uint8_t c, I &i) {
@@ -10615,26 +16181,73 @@ simdjson_inline bool is_digit(const uint8_t c) {
   return static_cast<uint8_t>(c - '0') <= 9;
 }

-simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
-  // we continue with the fiction that we have an integer. If the
-  // floating point number is representable as x * 10^z for some integer
-  // z that fits in 53 bits, then we will be able to convert back the
-  // the integer into a float in a lossless manner.
-  const uint8_t *const first_after_period = p;
+// Consumes a run of digits into i, eight at a time while eight are available,
+// then four, then one at a time. Overflow is deliberate: the caller counts the
+// digits and falls back when there are too many for a 64-bit mantissa.
+//
+// Long runs of digits are what a float-heavy document is made of: the fraction
+// of a binary64 printed to full precision is 15 to 17 digits, and reading those
+// one at a time is the single largest cost in parse_double(). Integer parts are
+// left alone: they are usually a handful of digits, and there the check for a
+// block of eight is wasted work.
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_fraction_digits(const uint8_t *&p, uint64_t &i) {
+#ifdef SIMDJSON_SWAR_NUMBER_PARSING
+#if SIMDJSON_SWAR_NUMBER_PARSING
+  while (is_made_of_eight_digits_fast(p)) {
+    i = i * 100000000 + parse_eight_digits_unrolled(p);
+    p += 8;
+  }
+  // A 4 to 7 digit remainder is the common case once the blocks of eight are
+  // gone: 15 fraction digits leave 7, and 17 leave 1 after two blocks.
+  if (is_made_of_four_digits_fast(p)) {
+    i = i * 10000 + parse_four_digits_unrolled(p);
+    p += 4;
+  }
+#endif // SIMDJSON_SWAR_NUMBER_PARSING
+#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
+  while (parse_digit(*p, i)) { p++; }
+}

+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_float_integer_digits(const uint8_t *&p, uint64_t &i,
+                                                bool &leading_zero) {
+  if (!parse_digit(*p, i)) { leading_zero = true; return; }
+  p++;
+  leading_zero = (i == 0);
+  while (parse_digit(*p, i)) { p++; }
+}
+
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_integer_digits(const uint8_t *&p, uint64_t &i) {
 #ifdef SIMDJSON_SWAR_NUMBER_PARSING
 #if SIMDJSON_SWAR_NUMBER_PARSING
-  // this helps if we have lots of decimals!
-  // this turns out to be frequent enough.
+  // Identifiers, timestamps and counters often have eight digits or more.
+  // Prior related work: jsonifier parses integers as eight-digit SWAR words
+  // (str_to_i.hpp, https://github.com/nihilai-collective/Jsonifier). This
+  // takes one such word with parse_eight_digits_unrolled, the routine
+  // simdjson uses for long fractions.
   if (is_made_of_eight_digits_fast(p)) {
     i = i * 100000000 + parse_eight_digits_unrolled(p);
     p += 8;
   }
+  const uint8_t *const swar_end = p + 8;
+  while (p < swar_end && is_made_of_four_digits_fast(p)) {
+    i = i * 10000 + parse_four_digits_unrolled(p);
+    p += 4;
+  }
 #endif // SIMDJSON_SWAR_NUMBER_PARSING
 #endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
-  // Unrolling the first digit makes a small difference on some implementations (e.g. westmere)
-  if (parse_digit(*p, i)) { ++p; }
   while (parse_digit(*p, i)) { p++; }
+}
+
+simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
+  // we continue with the fiction that we have an integer. If the
+  // floating point number is representable as x * 10^z for some integer
+  // z that fits in 53 bits, then we will be able to convert back the
+  // the integer into a float in a lossless manner.
+  const uint8_t *const first_after_period = p;
+  parse_fraction_digits(p, i);
   exponent = first_after_period - p;
   // Decimal without digits (123.) is illegal
   if (exponent == 0) {
@@ -10723,7 +16336,7 @@ simdjson_inline size_t significant_digits(const uint8_t * start_digits, size_t d
 } // unnamed namespace

 /** @private */
-static error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
+inline error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
   if (parse_float_fallback(src, answer)) {
     return SUCCESS;
   }
@@ -10806,15 +16419,17 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
   return SUCCESS;              // always succeeds
 }

-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept  { return false; }
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept  { return false; }
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept { return number_type::signed_integer; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline bool is_negative(const uint8_t *) noexcept  { return false; }
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t *) noexcept  { return false; }
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t *) noexcept { return number_type::signed_integer; }
 #else

 // parse the number at src
@@ -10845,7 +16460,21 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
   size_t digit_count = size_t(p - start_digits);
-  if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) { return INVALID_NUMBER(src); }
+  if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // By this point, we know that our input does not begin with a digit. We will attempt
+    // to handle NaN/Infinity.
+
+    double d;
+    if (compute_nan_inf(p, negative, d)) {
+      writer.append_double(d);
+      return SUCCESS;
+    }
+#endif
+
+    return INVALID_NUMBER(src);
+  }

   //
   // Handle floats if there is a . or e (or both)
@@ -10894,7 +16523,7 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
     // - Therefore, if the number is positive and lower than that, it's overflow.
     // - The value we are looking at is less than or equal to INT64_MAX.
     //
-    }  else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return INVALID_NUMBER(src); }
+    }  else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return BIGINT_NUMBER(src); }
   }

   // Write unsigned if it does not fit in a signed integer.
@@ -10993,7 +16622,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const u
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -11091,7 +16720,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_stri
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -11146,7 +16775,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uin
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -11232,7 +16861,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = src;
   uint64_t i = 0;
-  while (parse_digit(*src, i)) { src++; }
+  parse_integer_digits(src, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -11272,11 +16901,20 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   //
   uint64_t i = 0;
   const uint8_t *p = src;
-  p += parse_digit(*p, i);
-  bool leading_zero = (i == 0);
-  while (parse_digit(*p, i)) { p++; }
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
   // no integer digits, or 0123 (zero must be solo)
-  if ( p == src ) { return INCORRECT_TYPE; }
+  if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no loading digits, the number may be nan or infinity.
+    // Attempt to compute those, and return on success.
+    double d;
+    if (compute_nan_inf(p, negative, d)) { return d; }
+#endif
+
+    return INCORRECT_TYPE;
+  }
   if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }

   //
@@ -11287,9 +16925,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   if (simdjson_likely(*p == '.')) {
     p++;
     const uint8_t *start_decimal_digits = p;
-    if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
-    p++;
-    while (parse_digit(*p, i)) { p++; }
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
     exponent = -(p - start_decimal_digits);

     // Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -11338,6 +16975,97 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   return d;
 }

+// Parse a JSON number into a binary32 (float) value.
+//
+// This mirrors parse_double, but it rounds to binary32 directly instead of
+// rounding to binary64 and then to binary32: the latter is subject to double
+// rounding and would not always produce the nearest float.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * src) noexcept {
+  //
+  // Check for minus sign
+  //
+  bool negative = (*src == '-');
+  src += uint8_t(negative);
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, the number may be nan or infinity.
+    // Attempt to compute those, and return on success.
+    float f;
+    if (compute_nan_inf(p, negative, f)) { return f; }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = p-start_digits > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if (*p == 'e' || *p == 'E') {
+    p++;
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while (parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  float d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_32(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
 simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
   return (*src == '-');
 }
@@ -11490,11 +17218,25 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   //
   uint64_t i = 0;
   const uint8_t *p = src;
-  p += parse_digit(*p, i);
-  bool leading_zero = (i == 0);
-  while (parse_digit(*p, i)) { p++; }
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
   // no integer digits, or 0123 (zero must be solo)
-  if ( p == src ) { return INCORRECT_TYPE; }
+  if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, attempt to parse numbers that are either
+    // NaN or Infinity
+    if (atomparsing::is_valid_inf_in_string(src)) {
+      double inf = std::numeric_limits<double>::infinity();
+      return negative ? -inf : inf;
+    }
+
+    if (atomparsing::is_valid_nan_in_string(src)) {
+      return std::numeric_limits<double>::quiet_NaN();
+    }
+#endif
+
+    return INCORRECT_TYPE;
+  }
   if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }

   //
@@ -11505,9 +17247,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   if (simdjson_likely(*p == '.')) {
     p++;
     const uint8_t *start_decimal_digits = p;
-    if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
-    p++;
-    while (parse_digit(*p, i)) { p++; }
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
     exponent = -(p - start_decimal_digits);

     // Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -11556,6 +17297,99 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   return d;
 }

+// Parse a JSON number held inside a JSON string into a binary32 (float) value.
+// See parse_float for why we do not simply round parse_double_in_string.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * src) noexcept {
+  //
+  // Check for minus sign
+  //
+  bool negative = (*(src + 1) == '-');
+  src += uint8_t(negative) + 1;
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, attempt to parse numbers that are either
+    // NaN or Infinity
+    if (atomparsing::is_valid_inf_in_string(src)) {
+      float inf = std::numeric_limits<float>::infinity();
+      return negative ? -inf : inf;
+    }
+
+    if (atomparsing::is_valid_nan_in_string(src)) {
+      return std::numeric_limits<float>::quiet_NaN();
+    }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = p-start_digits > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if (*p == 'e' || *p == 'E') {
+    p++;
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while (parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if (*p != '"') { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  float d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_32(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
 } // unnamed namespace
 #endif // SIMDJSON_SKIPNUMBERPARSING

@@ -11749,6 +17583,9 @@ public:
 #endif // SIMDJSON_ARM64_IMPLEMENTATION_H
 /* end file simdjson/arm64/implementation.h */

+// defining SIMDJSON_GENERIC_JSON_STRUCTURAL_INDEXER_CUSTOM_BIT_INDEXER allows us to provide our own bit_indexer::write
+#define SIMDJSON_GENERIC_JSON_STRUCTURAL_INDEXER_CUSTOM_BIT_INDEXER
+
 /* including simdjson/arm64/begin.h: #include <simdjson/arm64/begin.h> */
 /* begin file simdjson/arm64/begin.h */
 /* defining SIMDJSON_IMPLEMENTATION to "arm64" */
@@ -11861,7 +17698,14 @@ simdjson_inline int leading_zeroes(uint64_t input_num) {

 /* result might be undefined when input_num is zero */
 simdjson_inline int count_ones(uint64_t input_num) {
+#if SIMDJSON_REGULAR_VISUAL_STUDIO
    return vaddv_u8(vcnt_u8(vcreate_u8(input_num)));
+#else
+   // if the system supports SVE or CSSC, __builtin_popcountll
+   // might be compiled to fewer single instructions. For CSSC,
+   // __builtin_popcountll is compiled to a single instruction.
+   return __builtin_popcountll(input_num);
+#endif// SIMDJSON_REGULAR_VISUAL_STUDIO
 }


@@ -11898,15 +17742,6 @@ simdjson_inline uint64_t zero_leading_bit(uint64_t rev_bits, int leading_zeroes)

 #endif

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2, uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
-  *result = value1 + value2;
-  return *result < value1;
-#else
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-#endif
-}

 } // unnamed namespace
 } // namespace arm64
@@ -12170,6 +18005,7 @@ namespace {
       return vget_lane_u64(
           vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(*this), 4)), 0);
     }
+    // Integer umaxv, not a float compare: flush-to-zero would treat a denormal as zero.
     simdjson_inline bool any() const { return vmaxvq_u32(vreinterpretq_u32_u8(*this)) != 0; }
   };

@@ -13615,6 +19451,296 @@ simdjson_inline uint32_t find_next_document_index(dom_parser_implementation &par
   return 0;
 }

+/**
+ * Sentinel value returned to indicate a document started but didn't fit
+ * (CAPACITY error), as opposed to 0 which means no document content found
+ * (EMPTY).
+ */
+constexpr uint32_t DOCUMENT_TOO_LARGE = UINT32_MAX;
+
+/**
+ * For RFC 7464 JSON text sequences, filter RS from structural indexes and
+ * find batch boundaries.
+ *
+ * In JSON sequence mode, RS (0x1E) marks the start of each JSON text.
+ * RS bytes appear in structural_indexes as they are classified as scalars.
+ * This function:
+ * 1. Scans structural_indexes to find and count RS positions
+ * 2. Filters RS out of structural_indexes in-place
+ * 3. Determines batch boundaries based on RS positions
+ *
+ * @param parser The parser with structural_indexes and buf.
+ * @param len The length of the current batch buffer.
+ * @param is_final True if this is the final batch (no more data coming).
+ * @param next_batch_start Output: offset where the next batch should start.
+ * @param scan_len Offset past which no value start may be derived. Equals len
+ *        unless stage 1 dropped a trailing unclosed string, whose bytes it
+ *        never validated.
+ * @return The number of structural indexes to keep (after RS filtering),
+ *         0 if no document content found (EMPTY),
+ *         or DOCUMENT_TOO_LARGE if a document started but didn't fit (CAPACITY).
+ */
+simdjson_inline uint32_t find_next_document_index_json_sequence(
+    dom_parser_implementation &parser,
+    size_t len,
+    bool is_final,
+    uint32_t &next_batch_start,
+    size_t scan_len) {
+  // Default: next batch starts at end of buffer
+  next_batch_start = uint32_t(len);
+
+  if (parser.n_structural_indexes == 0) { return 0; }
+
+  // Phase 1: Scan structural_indexes to find RS positions and handle them.
+  // RS marks the start of a JSON text. For objects/arrays, the '{' or '[' after RS
+  // is already in structural_indexes (it's an operator). For scalars like numbers,
+  // the digit following RS is NOT in structural_indexes because the scanner sees
+  // RS as a scalar, making the digit a scalar continuation, not a start.
+  // We must: (1) remove RS from structural_indexes, and (2) for scalars, add the
+  // actual value start position.
+  // The EOF sentinel: len, or where a discarded unclosed string starts.
+  const uint32_t sentinel = parser.structural_indexes[parser.n_structural_indexes];
+  uint32_t write_idx = 0;
+  uint32_t last_rs_pos = 0;
+  uint32_t rs_count = 0;
+
+  for (uint32_t read_idx = 0; read_idx < parser.n_structural_indexes; read_idx++) {
+    const uint32_t pos = parser.structural_indexes[read_idx];
+    if (parser.buf[pos] == 0x1E) {
+      // This is an RS character - find the actual JSON value start.
+      last_rs_pos = pos;
+      rs_count++;
+      // Skip past this RS and any whitespace *and any additional RSes*
+      // to locate the real value. Consecutive RSes are degenerate
+      // "empty records" per RFC 7464; we collapse them here. They do
+      // not always appear as separate entries in structural_indexes
+      // because the scanner groups runs of adjacent non-whitespace
+      // scalar bytes (including RS) into a single scalar start.
+      uint32_t value_start = pos + 1;
+      while (value_start < scan_len) {
+        const uint8_t c = parser.buf[value_start];
+        if (c == ' ' || c == '\t' || c == '\n' || c == '\r') {
+          value_start++;
+        } else if (c == 0x1E) {
+          // Collapsed empty record. Still count it so rs_count reflects
+          // the true number of record markers and last_rs_pos tracks
+          // the final one.
+          last_rs_pos = value_start;
+          rs_count++;
+          value_start++;
+        } else {
+          break;
+        }
+      }
+      // If the scanner emitted additional structurals inside the
+      // whitespace+RS run we just walked over (i.e., isolated RSes
+      // separated by whitespace), skip past them so we do not
+      // double-count or double-emit.
+      while (read_idx + 1 < parser.n_structural_indexes &&
+             parser.structural_indexes[read_idx + 1] < value_start) {
+        read_idx++;
+      }
+      // Check if the value start is an operator (always present in
+      // scanner structural_indexes) or a scalar-like start (which may
+      // be missing from structural_indexes and must be added here).
+      // Note: '"' is NOT always in structural_indexes. The scanner
+      // classifies '"' as a scalar character and emits it as a
+      // structural only when it is a *scalar start* (preceded by
+      // whitespace or an operator). When '"' immediately follows an
+      // RS (which the scanner also classifies as scalar), it is
+      // treated as a scalar continuation and not emitted - so we
+      // must add it here just like any other scalar value.
+      if (value_start < scan_len) {
+        const uint8_t c = parser.buf[value_start];
+        const bool is_operator =
+            (c == '{' || c == '}' || c == '[' || c == ']' ||
+             c == ':' || c == ',');
+        // If the next scanner structural is exactly at value_start,
+        // the scanner already emitted it (it followed whitespace) and
+        // we must not add a duplicate - a subsequent iteration will
+        // copy it into write_idx.
+        const bool already_emitted =
+            (read_idx + 1 < parser.n_structural_indexes &&
+             parser.structural_indexes[read_idx + 1] == value_start);
+        if (!is_operator && !already_emitted) {
+          // Scalar value (number/true/false/null/string) - add its
+          // position since scanner missed it.
+          parser.structural_indexes[write_idx++] = value_start;
+        }
+      }
+    } else {
+      // Not RS, copy to output
+      parser.structural_indexes[write_idx++] = pos;
+    }
+  }
+
+  // Update structural index count. Compaction left a stale index in the slot
+  // past the end: restore the EOF sentinel that stage 1 had planted there, which
+  // document_stream::truncated_bytes() reads after a final batch.
+  parser.n_structural_indexes = write_idx;
+  parser.structural_indexes[write_idx] = sentinel;
+
+  if (parser.n_structural_indexes == 0) {
+    // Only RS markers here: the last one opens a record continuing past the
+    // window, so that is where the next batch resumes.
+    if (!is_final && rs_count > 0) { next_batch_start = last_rs_pos; }
+    return 0;
+  }
+  if (rs_count == 0) {
+    // No RS found; for final batch, try generic boundary detection
+    return is_final ? find_next_document_index(parser) : 0;
+  }
+
+  // Phase 2: Determine batch boundaries based on RS positions
+
+  if (is_final) {
+    // Final batch: all documents are complete (last one ends at EOF).
+    // In json_sequence mode, RS markers define document boundaries, so all
+    // remaining structurals form complete documents. Return them all directly.
+    // (Calling find_next_document_index() would fail for scalar documents.)
+    return parser.n_structural_indexes;
+  }
+
+  // Partial batch: need to find complete documents only.
+  // A document starting at an RS is complete if there is another RS after it.
+  next_batch_start = last_rs_pos;
+
+  if (rs_count < 2) {
+    // Only one RS, so we have at most one document that may be incomplete.
+    // We cannot confirm it is complete without another RS.
+    // Return DOCUMENT_TOO_LARGE if content was found (write_idx > 0), 0 if only separators.
+    return (parser.n_structural_indexes > 0) ? DOCUMENT_TOO_LARGE : 0;
+  }
+
+  // We have at least 2 RS markers. The last complete document ends before last_rs_pos.
+
+  // Find the structural index cutoff: keep only structurals < last_rs_pos.
+  // Since we already filtered RS, all remaining structurals are valid.
+  // We iterate backward to find the last structural before last_rs_pos.
+  uint32_t keep_count = 0;
+  for (uint32_t i = parser.n_structural_indexes; i > 0; i--) {
+    if (parser.structural_indexes[i - 1] < last_rs_pos) {
+      keep_count = i;
+      break;
+    }
+  }
+
+  // No structurals before the last RS - no complete documents
+  if (keep_count == 0) { return 0; }
+
+  // All documents before the last RS are complete by definition (the next RS
+  // confirms their end). No need to call find_next_document_index() which
+  // would fail for scalar documents like `1` or `"hello"`.
+  return keep_count;
+}
+
+/**
+ * Filter comma-delimited documents by removing root-level commas from
+ * structural indexes.
+ *
+ * For comma-delimited format like `{...},{...},{...}`, we need to remove
+ * the commas that separate documents (depth 0) while preserving commas
+ * inside arrays and objects (depth > 0).
+ *
+ * After filtering, the structural indexes look like whitespace-delimited
+ * documents, so find_next_document_index() works unchanged.
+ *
+ * @param parser The parser with structural_indexes and buf.
+ * @param len The length of the current batch buffer.
+ * @param is_final True if this is the final batch (no more data coming).
+ * @param next_batch_start Output: offset where the next batch should start.
+ * @return The number of structural indexes to keep,
+ *         0 if no document content found (EMPTY),
+ *         or DOCUMENT_TOO_LARGE if a document started but didn't fit (CAPACITY).
+ */
+simdjson_inline uint32_t filter_comma_delimited(
+    dom_parser_implementation &parser,
+    size_t len,
+    bool is_final,
+    uint32_t &next_batch_start) {
+  // Default: next batch starts at end of buffer
+  next_batch_start = uint32_t(len);
+
+  if (parser.n_structural_indexes == 0) { return 0; }
+
+  // Track depth to identify root-level commas (depth 0)
+  int depth = 0;
+  // The EOF sentinel: len, or where a discarded unclosed string starts.
+  const uint32_t sentinel = parser.structural_indexes[parser.n_structural_indexes];
+  uint32_t write_idx = 0;
+  uint32_t last_root_comma_pos = 0;
+  uint32_t root_comma_count = 0;
+
+  for (uint32_t i = 0; i < parser.n_structural_indexes; i++) {
+    uint32_t idx = parser.structural_indexes[i];
+    uint8_t c = parser.buf[idx];
+
+    switch (c) {
+      case '{': case '[':
+        depth++;
+        parser.structural_indexes[write_idx++] = idx;
+        break;
+      case '}': case ']':
+        depth--;
+        parser.structural_indexes[write_idx++] = idx;
+        break;
+      case ',':
+        if (depth == 0) {
+          // Root-level comma = document boundary, skip it
+          last_root_comma_pos = idx;
+          root_comma_count++;
+          continue;
+        }
+        parser.structural_indexes[write_idx++] = idx;
+        break;
+      default:
+        // Colons, scalars, etc.
+        parser.structural_indexes[write_idx++] = idx;
+        break;
+    }
+  }
+
+  // Update structural index count. Compaction left a stale index in the slot
+  // past the end: restore the EOF sentinel that stage 1 had planted there, which
+  // document_stream::truncated_bytes() reads after a final batch.
+  parser.n_structural_indexes = write_idx;
+  parser.structural_indexes[write_idx] = sentinel;
+
+  if (parser.n_structural_indexes == 0) { return 0; }
+
+  if (is_final) {
+    // Final batch: use standard boundary detection on filtered indexes
+    return find_next_document_index(parser);
+  }
+
+  // Partial batch: need to find complete documents only.
+  // A document ending with a root comma is complete.
+  if (root_comma_count == 0) {
+    // No root commas found; we cannot confirm any document is complete.
+    // The whole batch might be one incomplete document.
+    // Return DOCUMENT_TOO_LARGE if content was found (write_idx > 0), 0 if only commas.
+    return (parser.n_structural_indexes > 0) ? DOCUMENT_TOO_LARGE : 0;
+  }
+
+  // We have at least one root comma. Documents before the last comma are complete.
+  next_batch_start = last_root_comma_pos + 1;
+
+  // Find the structural index cutoff: keep only structurals < last_root_comma_pos
+  uint32_t keep_count = 0;
+  for (uint32_t i = parser.n_structural_indexes; i > 0; i--) {
+    if (parser.structural_indexes[i - 1] < last_root_comma_pos) {
+      keep_count = i;
+      break;
+    }
+  }
+
+  if (keep_count == 0) { return 0; }
+
+  // Use standard boundary detection on the complete portion
+  parser.n_structural_indexes = keep_count;
+  return find_next_document_index(parser);
+}
+
 } // namespace stage1
 } // unnamed namespace
 } // namespace arm64
@@ -14048,7 +20174,6 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
         return EMPTY;
       }
     }
-
     parser.n_structural_indexes = new_structural_indexes;
   } else if (partial == stage1_mode::streaming_final) {
     if(have_unclosed_string) { parser.n_structural_indexes--; }
@@ -14076,6 +20201,75 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
         // the trailing garbage.
         return EMPTY;
     }
+  } else if (partial == stage1_mode::json_sequence_partial) {
+    // RFC 7464: use RS positions for batch boundaries
+    // A discarded unclosed string also caps how far the filter may scan.
+    size_t scan_len = len;
+    if(have_unclosed_string) {
+      parser.n_structural_indexes--;
+      if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return CAPACITY; }
+      scan_len = parser.structural_indexes[parser.n_structural_indexes];
+    }
+    uint32_t next_batch_start = uint32_t(len);
+    auto new_structural_indexes = find_next_document_index_json_sequence(parser, len, false, next_batch_start, scan_len);
+    if (new_structural_indexes == DOCUMENT_TOO_LARGE) {
+      return CAPACITY;
+    }
+    if (new_structural_indexes == 0) {
+      // An EMPTY batch must still advance next_batch_start, or document_stream
+      // re-parses the same bytes forever. CAPACITY when it cannot.
+      if (next_batch_start == 0) { return CAPACITY; }
+      parser.n_structural_indexes = 0;
+      parser.structural_indexes[0] = next_batch_start;
+      return EMPTY;
+    }
+    parser.n_structural_indexes = new_structural_indexes;
+    parser.structural_indexes[parser.n_structural_indexes] = next_batch_start;
+  } else if (partial == stage1_mode::json_sequence_final) {
+    // RFC 7464: final batch, last document extends to EOF
+    // As above: a discarded unclosed string caps how far the filter may scan.
+    size_t scan_len = len;
+    if(have_unclosed_string) {
+      parser.n_structural_indexes--;
+      scan_len = parser.structural_indexes[parser.n_structural_indexes];
+    }
+    uint32_t next_batch_start = uint32_t(len);
+    parser.n_structural_indexes = find_next_document_index_json_sequence(parser, len, true, next_batch_start, scan_len);
+    // The filter compacted structural_indexes in place and restored the EOF
+    // sentinel past the compacted end, so the copy below is either the start
+    // of a truncated document or len, as in streaming_final.
+    parser.structural_indexes[parser.n_structural_indexes + 1] = parser.structural_indexes[parser.n_structural_indexes];
+    parser.structural_indexes[parser.n_structural_indexes] = uint32_t(len);
+    if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return EMPTY; }
+  } else if (partial == stage1_mode::comma_delimited_partial) {
+    // Comma-delimited: filter root-level commas, use comma positions for batch boundaries
+    if(have_unclosed_string) {
+      parser.n_structural_indexes--;
+      if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return CAPACITY; }
+    }
+    uint32_t next_batch_start = uint32_t(len);
+    auto new_structural_indexes = filter_comma_delimited(parser, len, false, next_batch_start);
+    if (new_structural_indexes == DOCUMENT_TOO_LARGE) {
+      return CAPACITY;
+    }
+    if (new_structural_indexes == 0) {
+      // An EMPTY batch must still advance next_batch_start, or document_stream
+      // re-parses the same bytes forever. CAPACITY when it cannot.
+      if (next_batch_start == 0) { return CAPACITY; }
+      parser.n_structural_indexes = 0;
+      parser.structural_indexes[0] = next_batch_start;
+      return EMPTY;
+    }
+    parser.n_structural_indexes = new_structural_indexes;
+    parser.structural_indexes[parser.n_structural_indexes] = next_batch_start;
+  } else if (partial == stage1_mode::comma_delimited_final) {
+    // Comma-delimited: final batch, last document extends to EOF
+    if(have_unclosed_string) { parser.n_structural_indexes--; }
+    uint32_t next_batch_start = uint32_t(len);
+    parser.n_structural_indexes = filter_comma_delimited(parser, len, true, next_batch_start);
+    parser.structural_indexes[parser.n_structural_indexes + 1] = parser.structural_indexes[parser.n_structural_indexes];
+    parser.structural_indexes[parser.n_structural_indexes] = uint32_t(len);
+    if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return EMPTY; }
   }
   checker.check_eof();
   return checker.errors();
@@ -14158,7 +20352,6 @@ namespace {
 namespace stage2 {

 class json_iterator;
-class structural_iterator;
 struct tape_builder;
 struct tape_writer;

@@ -14455,7 +20648,7 @@ public:
    *
    * - increment_count(iter) - each time a value is found in an array or object.
    */
-  template<bool STREAMING, typename V>
+  template<bool STREAMING, bool UNPADDED, typename V>
   simdjson_warn_unused simdjson_inline error_code walk_document(V &visitor) noexcept;

   /**
@@ -14472,6 +20665,7 @@ public:
    *
    * They may include invalid JSON as well (such as `1.2.3` or `ture`).
    */
+  template<bool UNPADDED>
   simdjson_inline const uint8_t *peek() const noexcept;
   /**
    * Advance to the next token.
@@ -14480,6 +20674,7 @@ public:
    *
    * They may include invalid JSON as well (such as `1.2.3` or `ture`).
    */
+  template<bool UNPADDED>
   simdjson_inline const uint8_t *advance() noexcept;
   /**
    * Get the remaining length of the document, from the start of the current token.
@@ -14528,7 +20723,7 @@ public:
   simdjson_warn_unused simdjson_inline error_code visit_primitive(V &visitor, const uint8_t *value) noexcept;
 };

-template<bool STREAMING, typename V>
+template<bool STREAMING, bool UNPADDED, typename V>
 simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &visitor) noexcept {
   logger::log_start();

@@ -14543,7 +20738,7 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &
   // Read first value
   //
   {
-    auto value = advance();
+    auto value = advance<UNPADDED>();

     // Make sure the outer object or array is closed before continuing; otherwise, there are ways we
     // could get into memory corruption. See https://github.com/simdjson/simdjson/issues/906
@@ -14555,8 +20750,8 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &
     }

     switch (*value) {
-      case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
-      case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+      case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+      case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
       default: SIMDJSON_TRY( visitor.visit_root_primitive(*this, value) ); break;
     }
   }
@@ -14573,29 +20768,29 @@ object_begin:
   SIMDJSON_TRY( visitor.visit_object_start(*this) );

   {
-    auto key = advance();
+    auto key = advance<UNPADDED>();
     if (*key != '"') { log_error("Object does not start with a key"); return TAPE_ERROR; }
     SIMDJSON_TRY( visitor.increment_count(*this) );
     SIMDJSON_TRY( visitor.visit_key(*this, key) );
   }

 object_field:
-  if (simdjson_unlikely( *advance() != ':' )) { log_error("Missing colon after key in object"); return TAPE_ERROR; }
+  if (simdjson_unlikely( *advance<UNPADDED>() != ':' )) { log_error("Missing colon after key in object"); return TAPE_ERROR; }
   {
-    auto value = advance();
+    auto value = advance<UNPADDED>();
     switch (*value) {
-      case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
-      case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+      case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+      case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
       default: SIMDJSON_TRY( visitor.visit_primitive(*this, value) ); break;
     }
   }

 object_continue:
-  switch (*advance()) {
+  switch (*advance<UNPADDED>()) {
     case ',':
       SIMDJSON_TRY( visitor.increment_count(*this) );
       {
-        auto key = advance();
+        auto key = advance<UNPADDED>();
         if (simdjson_unlikely( *key != '"' )) { log_error("Key string missing at beginning of field in object"); return TAPE_ERROR; }
         SIMDJSON_TRY( visitor.visit_key(*this, key) );
       }
@@ -14623,16 +20818,16 @@ array_begin:

 array_value:
   {
-    auto value = advance();
+    auto value = advance<UNPADDED>();
     switch (*value) {
-      case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
-      case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+      case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+      case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
       default: SIMDJSON_TRY( visitor.visit_primitive(*this, value) ); break;
     }
   }

 array_continue:
-  switch (*advance()) {
+  switch (*advance<UNPADDED>()) {
     case ',': SIMDJSON_TRY( visitor.increment_count(*this) ); goto array_value;
     case ']': log_end_value("array"); SIMDJSON_TRY( visitor.visit_array_end(*this) ); goto scope_end;
     default: log_error("Missing comma between array values"); return TAPE_ERROR;
@@ -14660,11 +20855,29 @@ simdjson_inline json_iterator::json_iterator(dom_parser_implementation &_dom_par
     dom_parser{_dom_parser} {
 }

+// Stage 1 leaves a sentinel index of value `len`; reading buf[len] requires
+// padding. For UNPADDED, return a dummy so walk_document can fail cleanly (#2815).
+template<bool UNPADDED>
 simdjson_inline const uint8_t *json_iterator::peek() const noexcept {
-  return &buf[*(next_structural)];
+  const uint32_t idx = *(next_structural);
+  SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+    if (simdjson_unlikely(idx >= dom_parser.len)) {
+      static constexpr uint8_t unpadded_eof_sentinel = 0;
+      return &unpadded_eof_sentinel;
+    }
+  }
+  return &buf[idx];
 }
+template<bool UNPADDED>
 simdjson_inline const uint8_t *json_iterator::advance() noexcept {
-  return &buf[*(next_structural++)];
+  const uint32_t idx = *(next_structural++);
+  SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+    if (simdjson_unlikely(idx >= dom_parser.len)) {
+      static constexpr uint8_t unpadded_eof_sentinel = 0;
+      return &unpadded_eof_sentinel;
+    }
+  }
+  return &buf[idx];
 }
 simdjson_inline size_t json_iterator::remaining_len() const noexcept {
   return dom_parser.len - *(next_structural-1);
@@ -14704,7 +20917,20 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::visit_root_primit
     case '"': return visitor.visit_root_string(*this, value);
     case 't': return visitor.visit_root_true_atom(*this, value);
     case 'f': return visitor.visit_root_false_atom(*this, value);
+#if SIMDJSON_ENABLE_NAN_INF
+    case 'n': {
+      auto err = visitor.visit_root_null_atom(*this, value);
+      if (err == SUCCESS) { return err; }
+      // propagate the error value returned by a bad 'null' atom if parsing 'nan' fails
+      return visitor.visit_root_nan_atom(*this, value, err);
+    }
+    // 'N' isn't a canonically recognized atom, so we return a TAPE_ERROR if failure occurs
+    case 'N': return visitor.visit_root_nan_atom(*this, value, TAPE_ERROR);
+    case 'i':
+    case 'I': return visitor.visit_root_inf_atom(*this, value);
+#else
     case 'n': return visitor.visit_root_null_atom(*this, value);
+#endif
     case '-':
     case '0': case '1': case '2': case '3': case '4':
     case '5': case '6': case '7': case '8': case '9':
@@ -14726,7 +20952,20 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::visit_primitive(V
   switch (*value) {
     case 't': return visitor.visit_true_atom(*this, value);
     case 'f': return visitor.visit_false_atom(*this, value);
+#if SIMDJSON_ENABLE_NAN_INF
+    case 'n': {
+      auto err = visitor.visit_null_atom(*this, value);
+      if (err == SUCCESS) { return err; }
+      // propagate the error value returned by a bad 'null' atom if parsing 'nan' fails
+      return visitor.visit_nan_atom(*this, value, err);
+    }
+    // 'N' isn't a canonically recognized atom, so we return a TAPE_ERROR if failure occurs
+    case 'N': return visitor.visit_nan_atom(*this, value, TAPE_ERROR);
+    case 'i':
+    case 'I': return visitor.visit_inf_atom(*this, value);
+#else
     case 'n': return visitor.visit_null_atom(*this, value);
+#endif
     default:
       log_error("Non-value found when value was expected!");
       return TAPE_ERROR;
@@ -14912,9 +21151,13 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_string(const uint8_t *src, u
            within the unicode codepoint handling code. */
         src += bs_dist;
         dst += bs_dist;
-        if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
-          return nullptr;
-        }
+        // Decode adjacent Unicode escapes without returning to the
+        // quote-and-backslash scanner between code points.
+        do {
+          if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
+            return nullptr;
+          }
+        } while (src[0] == '\\' && src[1] == 'u');
       } else {
         /* simple 1:1 conversion. Will eat bs_dist+2 characters in input and
          * write bs_dist+1 characters to output
@@ -14937,6 +21180,77 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_string(const uint8_t *src, u
   }
 }

+/**
+ * Bounds-safe variant of parse_string for input buffers that are NOT padded to
+ * len + SIMDJSON_PADDING bytes. `buf_end` is one past the last readable input
+ * byte (buf + len). It behaves exactly like parse_string while we are at least
+ * SIMDJSON_PADDING bytes away from buf_end (so every speculative SIMD read stays
+ * in bounds); once within the final SIMDJSON_PADDING bytes it copies the few
+ * remaining bytes into a space-padded scratch buffer and finishes with the
+ * regular parse_string. This keeps the delicate escape/Unicode handling in one
+ * place (the proven parse_string) rather than duplicating it.
+ *
+ * Correctness relies on stage 1 having validated the string, i.e. there is an
+ * unescaped closing quote within [src, buf_end); that quote is therefore inside
+ * the copied scratch, so parse_string finds it without running off the scratch.
+ */
+simdjson_warn_unused simdjson_inline uint8_t *parse_string_safe(const uint8_t *src, uint8_t *dst, bool allow_replacement, const uint8_t *buf_end) {
+  // Far from the end: identical to parse_string's loop. The guard uses
+  // SIMDJSON_PADDING (>= BYTES_PROCESSED) so copy_and_find never reads past
+  // buf_end; escape/Unicode look-aheads read within the string (before the
+  // closing quote, which is < buf_end), so they are in bounds here too.
+  // We add margin (+12) for handle_unicode_codepoint's worst-case lookahead:
+  // after seeing a high surrogate, it does hex_to_u32_nocheck on the immediate
+  // following bytes (+6 from the '\'), then (if it sees \u) another
+  // hex_to_u32_nocheck at +8..+11 relative to the backslash that started the
+  // escape. With bs_dist up to BYTES_PROCESSED-1 this reaches +11 from the
+  // chunk start. The +12 margin ensures that even on kernels where
+  // BYTES_PROCESSED == SIMDJSON_PADDING (e.g. icelake) the 4-byte read stays
+  // in-bounds. The scratch fallback (3*PAD) is already safe.
+  while (src + SIMDJSON_PADDING + 12 <= buf_end) {
+    auto b = backslash_and_quote{};
+    auto bs_quote = b.copy_and_find(src, dst);
+    if (bs_quote.has_quote_first()) {
+      return dst + bs_quote.quote_index();
+    }
+    if (bs_quote.has_backslash()) {
+      auto bs_dist = bs_quote.backslash_index();
+      uint8_t escape_char = src[bs_dist + 1];
+      if (escape_char == 'u') {
+        src += bs_dist;
+        dst += bs_dist;
+        if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
+          return nullptr;
+        }
+      } else {
+        uint8_t escape_result = escape_map[escape_char];
+        if (escape_result == 0u) {
+          return nullptr;
+        }
+        dst[bs_dist] = escape_result;
+        src += bs_dist + 2;
+        dst += bs_dist + 1;
+      }
+    } else {
+      src += backslash_and_quote::BYTES_PROCESSED;
+      dst += backslash_and_quote::BYTES_PROCESSED;
+    }
+  }
+  // Within the final SIMDJSON_PADDING bytes: copy what remains into a
+  // space-padded scratch (spaces are neither quote nor backslash, so they do not
+  // disturb matching) and let the regular parser finish from there. The closing
+  // quote is within `remaining` (< SIMDJSON_PADDING), so parse_string finds it in
+  // the chunk starting at some offset <= remaining and reads at most
+  // BYTES_PROCESSED (<= SIMDJSON_PADDING) further -- i.e. under 2*SIMDJSON_PADDING.
+  // We size at 3x for a comfortable margin (the unicode look-ahead reads a few
+  // extra bytes past an escape).
+  uint8_t scratch[SIMDJSON_PADDING * 3];
+  const size_t remaining = size_t(buf_end - src); // < SIMDJSON_PADDING
+  std::memset(scratch, ' ', sizeof(scratch));
+  std::memcpy(scratch, src, remaining);
+  return parse_string(scratch, dst, allow_replacement);
+}
+
 simdjson_warn_unused simdjson_inline uint8_t *parse_wobbly_string(const uint8_t *src, uint8_t *dst) {
   // It is not ideal that this function is nearly identical to parse_string.
   while (1) {
@@ -14991,73 +21305,6 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_wobbly_string(const uint8_t

 #endif // SIMDJSON_SRC_GENERIC_STAGE2_STRINGPARSING_H
 /* end file generic/stage2/stringparsing.h for arm64 */
-/* including generic/stage2/structural_iterator.h for arm64: #include <generic/stage2/structural_iterator.h> */
-/* begin file generic/stage2/structural_iterator.h for arm64 */
-#ifndef SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H
-
-/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
-/* amalgamation skipped (editor-only): #define SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H */
-/* amalgamation skipped (editor-only): #include <generic/stage2/base.h> */
-/* amalgamation skipped (editor-only): #include <simdjson/generic/dom_parser_implementation.h> */
-/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
-
-namespace simdjson {
-namespace arm64 {
-namespace {
-namespace stage2 {
-
-class structural_iterator {
-public:
-  const uint8_t* const buf;
-  uint32_t *next_structural;
-  dom_parser_implementation &dom_parser;
-
-  // Start a structural
-  simdjson_inline structural_iterator(dom_parser_implementation &_dom_parser, size_t start_structural_index)
-    : buf{_dom_parser.buf},
-      next_structural{&_dom_parser.structural_indexes[start_structural_index]},
-      dom_parser{_dom_parser} {
-  }
-  // Get the buffer position of the current structural character
-  simdjson_inline const uint8_t* current() {
-    return &buf[*(next_structural-1)];
-  }
-  // Get the current structural character
-  simdjson_inline char current_char() {
-    return buf[*(next_structural-1)];
-  }
-  // Get the next structural character without advancing
-  simdjson_inline char peek_next_char() {
-    return buf[*next_structural];
-  }
-  simdjson_inline const uint8_t* peek() {
-    return &buf[*next_structural];
-  }
-  simdjson_inline const uint8_t* advance() {
-    return &buf[*(next_structural++)];
-  }
-  simdjson_inline char advance_char() {
-    return buf[*(next_structural++)];
-  }
-  simdjson_inline size_t remaining_len() {
-    return dom_parser.len - *(next_structural-1);
-  }
-
-  simdjson_inline bool at_end() {
-    return next_structural == &dom_parser.structural_indexes[dom_parser.n_structural_indexes];
-  }
-  simdjson_inline bool at_beginning() {
-    return next_structural == dom_parser.structural_indexes.get();
-  }
-};
-
-} // namespace stage2
-} // unnamed namespace
-} // namespace arm64
-} // namespace simdjson
-
-#endif // SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H
-/* end file generic/stage2/structural_iterator.h for arm64 */
 /* including generic/stage2/tape_builder.h for arm64: #include <generic/stage2/tape_builder.h> */
 /* begin file generic/stage2/tape_builder.h for arm64 */
 #ifndef SIMDJSON_SRC_GENERIC_STAGE2_TAPE_BUILDER_H
@@ -15080,12 +21327,8 @@ namespace arm64 {
 namespace {
 namespace stage2 {

-struct tape_builder {
-  template<bool STREAMING>
-  simdjson_warn_unused static simdjson_inline error_code parse_document(
-    dom_parser_implementation &dom_parser,
-    dom::document &doc) noexcept;
-
+template <bool UNPADDED>
+struct tape_builder_impl {
   /** Called when a non-empty document starts. */
   simdjson_warn_unused simdjson_inline error_code visit_document_start(json_iterator &iter) noexcept;
   /** Called when a non-empty document ends without error. */
@@ -15138,88 +21381,130 @@ struct tape_builder {
   simdjson_warn_unused simdjson_inline error_code visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept;
   simdjson_warn_unused simdjson_inline error_code visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept;

+#if SIMDJSON_ENABLE_NAN_INF
+  simdjson_warn_unused simdjson_inline error_code visit_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept;
+  simdjson_warn_unused simdjson_inline error_code visit_root_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept;
+  // Attempts to parse 'inf' or 'infinity' (case insensitive). Because neither are canonical atoms,
+  // this returns a tape error on failure.
+  simdjson_warn_unused simdjson_inline error_code visit_inf_atom(json_iterator &iter, const uint8_t *value) noexcept;
+  simdjson_warn_unused simdjson_inline error_code visit_root_inf_atom(json_iterator &iter, const uint8_t *value) noexcept;
+#endif
+
   /** Called each time a new field or element in an array or object is found. */
   simdjson_warn_unused simdjson_inline error_code increment_count(json_iterator &iter) noexcept;

   /** Next location to write to tape */
   tape_writer tape;
+public:
+  simdjson_inline tape_builder_impl(dom::document &doc) noexcept;
 private:
   /** Next write location in the string buf for stage 2 parsing */
   uint8_t *current_string_buf_loc;

-  simdjson_inline tape_builder(dom::document &doc) noexcept;
-
   simdjson_inline uint32_t next_tape_index(json_iterator &iter) const noexcept;
   simdjson_inline void start_container(json_iterator &iter) noexcept;
   simdjson_warn_unused simdjson_inline error_code end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept;
   simdjson_warn_unused simdjson_inline error_code empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept;
   simdjson_inline uint8_t *on_start_string(json_iterator &iter) noexcept;
   simdjson_inline void on_end_string(uint8_t *dst) noexcept;
-}; // struct tape_builder
+}; // struct tape_builder_impl

-template<bool STREAMING>
-simdjson_warn_unused simdjson_inline error_code tape_builder::parse_document(
-    dom_parser_implementation &dom_parser,
-    dom::document &doc) noexcept {
-  dom_parser.doc = &doc;
-  json_iterator iter(dom_parser, STREAMING ? dom_parser.next_structural_index : 0);
-  tape_builder builder(doc);
-  return iter.walk_document<STREAMING>(builder);
-}
+// Thin, non-templated entry so each architecture's stage2() keeps calling
+// tape_builder::parse_document<STREAMING> unchanged. It chooses the bounds-safe
+// (unpadded) or the regular (padded) tape_builder_impl ONCE per document, so the
+// choice is a compile-time constant inside the walk: the padded path carries no
+// extra branch or load (see tape_builder_impl::visit_string).
+struct tape_builder {
+  template<bool STREAMING>
+  simdjson_warn_unused static simdjson_inline error_code parse_document(
+      dom_parser_implementation &dom_parser, dom::document &doc) noexcept {
+    dom_parser.doc = &doc;
+    json_iterator iter(dom_parser, STREAMING ? dom_parser.next_structural_index : 0);
+    if (dom_parser._unpadded) {
+      tape_builder_impl<true> builder(doc);
+      return iter.walk_document<STREAMING, true>(builder);
+    } else {
+      tape_builder_impl<false> builder(doc);
+      return iter.walk_document<STREAMING, false>(builder);
+    }
+  }
+};

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_primitive(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_primitive(json_iterator &iter, const uint8_t *value) noexcept {
   return iter.visit_root_primitive(*this, value);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_primitive(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_primitive(json_iterator &iter, const uint8_t *value) noexcept {
   return iter.visit_primitive(*this, value);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_empty_object(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_empty_object(json_iterator &iter) noexcept {
   return empty_container(iter, internal::tape_type::START_OBJECT, internal::tape_type::END_OBJECT);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_empty_array(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_empty_array(json_iterator &iter) noexcept {
   return empty_container(iter, internal::tape_type::START_ARRAY, internal::tape_type::END_ARRAY);
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_document_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_document_start(json_iterator &iter) noexcept {
   start_container(iter);
   return SUCCESS;
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_object_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_object_start(json_iterator &iter) noexcept {
   start_container(iter);
   return SUCCESS;
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_array_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_array_start(json_iterator &iter) noexcept {
   start_container(iter);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_object_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_object_end(json_iterator &iter) noexcept {
   return end_container(iter, internal::tape_type::START_OBJECT, internal::tape_type::END_OBJECT);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_array_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_array_end(json_iterator &iter) noexcept {
   return end_container(iter, internal::tape_type::START_ARRAY, internal::tape_type::END_ARRAY);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_document_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_document_end(json_iterator &iter) noexcept {
   constexpr uint32_t start_tape_index = 0;
   tape.append(start_tape_index, internal::tape_type::ROOT);
   tape_writer::write(iter.dom_parser.doc->tape[start_tape_index], next_tape_index(iter), internal::tape_type::ROOT);
   return SUCCESS;
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_key(json_iterator &iter, const uint8_t *key) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_key(json_iterator &iter, const uint8_t *key) noexcept {
   return visit_string(iter, key, true);
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::increment_count(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::increment_count(json_iterator &iter) noexcept {
   iter.dom_parser.open_containers[iter.depth].count++; // we have a key value pair in the object at parser.dom_parser.depth - 1
   return SUCCESS;
 }

-simdjson_inline tape_builder::tape_builder(dom::document &doc) noexcept : tape{doc.tape.get()}, current_string_buf_loc{doc.string_buf.get()} {}
+template <bool UNPADDED>
+simdjson_inline tape_builder_impl<UNPADDED>::tape_builder_impl(dom::document &doc) noexcept : tape{doc.tape.get()}, current_string_buf_loc{doc.string_buf.get()} {}

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_string(json_iterator &iter, const uint8_t *value, bool key) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_string(json_iterator &iter, const uint8_t *value, bool key) noexcept {
   iter.log_value(key ? "key" : "string");
   uint8_t *dst = on_start_string(iter);
-  dst = stringparsing::parse_string(value+1, dst, false); // We do not allow replacement when the escape characters are invalid.
+  // We do not allow replacement when the escape characters are invalid.
+  // UNPADDED is a compile-time constant chosen once per document by
+  // tape_builder::parse_document, so the padded build instantiates only the
+  // plain parse_string call below -- no runtime branch and no flag load.
+  SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+    dst = stringparsing::parse_string_safe(value+1, dst, false, iter.buf + iter.dom_parser.len);
+  } else {
+    dst = stringparsing::parse_string(value+1, dst, false);
+  }
   if (dst == nullptr) {
     iter.log_error("Invalid escape in string");
     return STRING_ERROR;
@@ -15228,27 +21513,48 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_string(json_
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_string(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_string(json_iterator &iter, const uint8_t *value) noexcept {
   return visit_string(iter, value);
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_number(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_number(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("number");
-  error_code err = numberparsing::parse_number(value, tape);
+  const uint8_t *num = value;
+  std::unique_ptr<uint8_t[]> copy{}; // keeps a padded copy of the tail alive when used
+  SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+    // numberparsing reads ahead in 8-byte blocks for floats
+    // (is_made_of_eight_digits_fast reads up to 7 bytes past the digits), so a
+    // number whose digits reach the final bytes of an unpadded buffer would read
+    // past it. *(next_structural) is the offset of the token following this
+    // number, hence an upper bound on where the digits end; when that is within
+    // SIMDJSON_PADDING of the end we parse from a space-padded copy of the tail
+    // (mirroring visit_root_number). This fires only for numbers near the end.
+    if (simdjson_unlikely(*(iter.next_structural) + SIMDJSON_PADDING > iter.dom_parser.len)) {
+      const size_t rl = iter.remaining_len(); // bytes from `value` to the end of the document
+      copy.reset(new (std::nothrow) uint8_t[rl + SIMDJSON_PADDING]);
+      if (copy.get() == nullptr) { return MEMALLOC; }
+      std::memcpy(copy.get(), value, rl);
+      std::memset(copy.get() + rl, ' ', SIMDJSON_PADDING);
+      num = copy.get();
+    }
+  }
+  error_code err = numberparsing::parse_number(num, tape);
   if (simdjson_unlikely(err == BIGINT_ERROR &&
       iter.dom_parser._number_as_string)) {
     // Write big integer to string buffer using the same format as strings.
     // Scan digits the same way parse_number does (skip optional '-', then digits).
-    const uint8_t *p = value;
+    const uint8_t *p = num;
     if (*p == '-') p++;
     while (numberparsing::is_digit(*p)) p++;
     // The digit run must be terminated by a structural or whitespace character; otherwise the
     // token is malformed (e.g. "123456789123456789123x").
     if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
-    size_t len = size_t(p - value);
+    size_t len = size_t(p - num);
     tape.append(current_string_buf_loc - iter.dom_parser.doc->string_buf.get(), internal::tape_type::BIGINT);
     uint8_t *dst = current_string_buf_loc + sizeof(uint32_t);
-    memcpy(dst, value, len);
+    memcpy(dst, num, len);
     dst += len;
     on_end_string(dst);
     return SUCCESS;
@@ -15256,7 +21562,8 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_number(json_
   return err;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_number(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_number(json_iterator &iter, const uint8_t *value) noexcept {
   //
   // We need to make a copy to make sure that the string is space terminated.
   // This is not about padding the input, which should already padded up
@@ -15270,76 +21577,142 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_number(
   // practice unless you are in the strange scenario where you have many JSON
   // documents made of single atoms.
   //
-  std::unique_ptr<uint8_t[]>copy(new (std::nothrow) uint8_t[iter.remaining_len() + SIMDJSON_PADDING]);
+  // In a stream, the input goes on with other documents: copy up to the next
+  // structural only, not to the end of the batch.
+  const size_t len = (std::min)(iter.remaining_len(), size_t(*iter.next_structural) - size_t(*(iter.next_structural - 1)));
+  std::unique_ptr<uint8_t[]>copy(new (std::nothrow) uint8_t[len + SIMDJSON_PADDING]);
   if (copy.get() == nullptr) { return MEMALLOC; }
-  std::memcpy(copy.get(), value, iter.remaining_len());
-  std::memset(copy.get() + iter.remaining_len(), ' ', SIMDJSON_PADDING);
+  std::memcpy(copy.get(), value, len);
+  std::memset(copy.get() + len, ' ', SIMDJSON_PADDING);
   error_code error = visit_number(iter, copy.get());
   return error;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("true");
-  if (!atomparsing::is_valid_true_atom(value)) { return T_ATOM_ERROR; }
+  // The non-length-aware validator reads a fixed 5 bytes; a malformed/truncated
+  // token at the very end of an unpadded buffer would over-read. Use the
+  // length-aware form there (the root variant already does this).
+  const bool ok = UNPADDED ? atomparsing::is_valid_true_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_true_atom(value);
+  if (!ok) { return T_ATOM_ERROR; }
   tape.append(0, internal::tape_type::TRUE_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("true");
   if (!atomparsing::is_valid_true_atom(value, iter.remaining_len())) { return T_ATOM_ERROR; }
   tape.append(0, internal::tape_type::TRUE_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("false");
-  if (!atomparsing::is_valid_false_atom(value)) { return F_ATOM_ERROR; }
+  const bool ok = UNPADDED ? atomparsing::is_valid_false_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_false_atom(value);
+  if (!ok) { return F_ATOM_ERROR; }
   tape.append(0, internal::tape_type::FALSE_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("false");
   if (!atomparsing::is_valid_false_atom(value, iter.remaining_len())) { return F_ATOM_ERROR; }
   tape.append(0, internal::tape_type::FALSE_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("null");
-  if (!atomparsing::is_valid_null_atom(value)) { return N_ATOM_ERROR; }
+  const bool ok = UNPADDED ? atomparsing::is_valid_null_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_null_atom(value);
+  if (!ok) { return N_ATOM_ERROR; }
   tape.append(0, internal::tape_type::NULL_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("null");
   if (!atomparsing::is_valid_null_atom(value, iter.remaining_len())) { return N_ATOM_ERROR; }
   tape.append(0, internal::tape_type::NULL_VALUE);
   return SUCCESS;
 }

+#if SIMDJSON_ENABLE_NAN_INF
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept {
+  iter.log_value("nan");
+  // For unpadded input use the length-aware validator so the 'infinity'-style
+  // 8-byte compare cannot read past the buffer on a malformed token at the end.
+  const bool ok = UNPADDED ? atomparsing::is_valid_nan_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_nan_atom(value);
+  if (!ok) { return errc; }
+  tape.append_double(std::numeric_limits<double>::quiet_NaN());
+  return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept {
+  iter.log_value("nan");
+  if (!atomparsing::is_valid_nan_atom(value, iter.remaining_len())) { return errc; }
+  tape.append_double(std::numeric_limits<double>::quiet_NaN());
+  return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_inf_atom(json_iterator &iter, const uint8_t *value) noexcept {
+  iter.log_value("inf");
+  // Because 'inf' is an extension, non a canonical atom, a tape error should be returned on failure.
+  // For unpadded input use the length-aware validator so the 'infinity' 8-byte
+  // compare cannot read past the buffer on a malformed token at the end.
+  const bool ok = UNPADDED ? atomparsing::is_valid_inf_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_inf_atom(value);
+  if (!ok) { return TAPE_ERROR; }
+  tape.append_double(std::numeric_limits<double>::infinity());
+  return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_inf_atom(json_iterator &iter, const uint8_t *value) noexcept {
+  iter.log_value("inf");
+  // Because 'inf' is an extension, non a canonical atom, a tape error should be returned on failure
+  if (!atomparsing::is_valid_inf_atom(value, iter.remaining_len())) { return TAPE_ERROR; }
+  tape.append_double(std::numeric_limits<double>::infinity());
+  return SUCCESS;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
 // private:

-simdjson_inline uint32_t tape_builder::next_tape_index(json_iterator &iter) const noexcept {
+template <bool UNPADDED>
+simdjson_inline uint32_t tape_builder_impl<UNPADDED>::next_tape_index(json_iterator &iter) const noexcept {
   return uint32_t(tape.next_tape_loc - iter.dom_parser.doc->tape.get());
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
   auto start_index = next_tape_index(iter);
   tape.append(start_index+2, start);
   tape.append(start_index, end);
   return SUCCESS;
 }

-simdjson_inline void tape_builder::start_container(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_inline void tape_builder_impl<UNPADDED>::start_container(json_iterator &iter) noexcept {
   iter.dom_parser.open_containers[iter.depth].tape_index = next_tape_index(iter);
   iter.dom_parser.open_containers[iter.depth].count = 0;
   tape.skip(); // We don't actually *write* the start element until the end.
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
   // Write the ending tape element, pointing at the start location
   const uint32_t start_tape_index = iter.dom_parser.open_containers[iter.depth].tape_index;
   tape.append(start_tape_index, end);
@@ -15352,13 +21725,15 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::end_container(json
   return SUCCESS;
 }

-simdjson_inline uint8_t *tape_builder::on_start_string(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_inline uint8_t *tape_builder_impl<UNPADDED>::on_start_string(json_iterator &iter) noexcept {
   // we advance the point, accounting for the fact that we have a NULL termination
   tape.append(current_string_buf_loc - iter.dom_parser.doc->string_buf.get(), internal::tape_type::STRING);
   return current_string_buf_loc + sizeof(uint32_t);
 }

-simdjson_inline void tape_builder::on_end_string(uint8_t *dst) noexcept {
+template <bool UNPADDED>
+simdjson_inline void tape_builder_impl<UNPADDED>::on_end_string(uint8_t *dst) noexcept {
   uint32_t str_length = uint32_t(dst - (current_string_buf_loc + sizeof(uint32_t)));
   // TODO check for overflow in case someone has a crazy string (>=4GB?)
   // But only add the overflow check when the document itself exceeds 4GB
@@ -15379,6 +21754,38 @@ simdjson_inline void tape_builder::on_end_string(uint8_t *dst) noexcept {
 /* end file generic/stage2/tape_builder.h for arm64 */
 /* end file generic/stage2/amalgamated.h for arm64 */

+#undef SIMDJSON_GENERIC_JSON_STRUCTURAL_INDEXER_CUSTOM_BIT_INDEXER
+
+namespace simdjson { namespace arm64 { namespace { namespace stage1 {
+
+// The generic bit_indexer::write emits the structural indexes in groups of
+// four (up to 24), so a block with five set bits computes and stores eight
+// indexes, three of them wasted. Typical JSON has four to eight structural
+// characters per 64-byte block. This version writes the first four indexes
+// unconditionally and then continues by groups of two, which cuts the wasted
+// work on such blocks at the cost of one more branch for dense blocks. Both
+// versions fall back to the same scalar loop past 24 indexes.
+simdjson_inline void bit_indexer::write(uint32_t idx, uint64_t bits) {
+  if (bits == 0) { return; }
+
+  const int cnt = static_cast<int>(count_ones(bits));
+#if SIMDJSON_PREFER_REVERSE_BITS
+  bits = reverse_bits(bits);
+#endif
+  write_indexes<0, 4>(idx, bits);
+  if (simdjson_unlikely(4 < cnt)) {
+    write_indexes_stepped<4, 24, 2>(idx, bits, cnt);
+  }
+  if (simdjson_unlikely(24 < cnt)) {
+    for (int i = 24; i < cnt; ++i) {
+      write_index(idx, bits, i);
+    }
+  }
+  this->tail += cnt;
+}
+
+}}}} // namespace simdjson::arm64::(anonymous)::stage1
+
 //
 // Stage 1
 //
@@ -15404,55 +21811,43 @@ namespace {
 using namespace simd;

 simdjson_inline json_character_block json_character_block::classify(const simd::simd8x64<uint8_t>& in) {
-  // Functional programming causes trouble with Visual Studio.
-  // Keeping this version in comments since it is much nicer:
-  // auto v = in.map<uint8_t>([&](simd8<uint8_t> chunk) {
-  //  auto nib_lo = chunk & 0xf;
-  //  auto nib_hi = chunk.shr<4>();
-  //  auto shuf_lo = nib_lo.lookup_16<uint8_t>(16, 0, 0, 0, 0, 0, 0, 0, 0, 8, 12, 1, 2, 9, 0, 0);
-  //  auto shuf_hi = nib_hi.lookup_16<uint8_t>(8, 0, 18, 4, 0, 1, 0, 1, 0, 0, 0, 3, 2, 1, 0, 0);
-  //  return shuf_lo & shuf_hi;
-  // });
-  const simd8<uint8_t> table1(16, 0, 0, 0, 0, 0, 0, 0, 0, 8, 12, 1, 2, 9, 0, 0);
-  const simd8<uint8_t> table2(8, 0, 18, 4, 0, 1, 0, 1, 0, 0, 0, 3, 2, 1, 0, 0);
-
-  simd8x64<uint8_t> v(
-     (in.chunks[0] & 0xf).lookup_16(table1) & (in.chunks[0].shr<4>()).lookup_16(table2),
-     (in.chunks[1] & 0xf).lookup_16(table1) & (in.chunks[1].shr<4>()).lookup_16(table2),
-     (in.chunks[2] & 0xf).lookup_16(table1) & (in.chunks[2].shr<4>()).lookup_16(table2),
-     (in.chunks[3] & 0xf).lookup_16(table1) & (in.chunks[3].shr<4>()).lookup_16(table2)
+  const uint8x16_t op_table = simd8<uint8_t>(
+    0xff, 0, ',', ':', 0, '[', ']', '{', '}', 0, 0, 0, 0, 0, 0, 0
+  );
+  const uint8x16_t ws_table = simd8<uint8_t>(
+    0, 0, 0, 0, 0, 0, 0, 0, 0, 0xff, 0xff, 0, 0, 0xff, 0, 0
   );

+  const uint8x16_t d0_0 = in.chunks[0];
+  const uint8x16_t d0_1 = in.chunks[1];
+  const uint8x16_t d0_2 = in.chunks[2];
+  const uint8x16_t d0_3 = in.chunks[3];

-  // We compute whitespace and op separately. If the code later only use one or the
-  // other, given the fact that all functions are aggressively inlined, we can
-  // hope that useless computations will be omitted. This is namely case when
-  // minifying (we only need whitespace). *However* if we only need spaces,
-  // it is likely that we will still compute 'v' above with two lookup_16: one
-  // could do it a bit cheaper. This is in contrast with the x64 implementations
-  // where we can, efficiently, do the white space and structural matching
-  // separately. One reason for this difference is that on ARM NEON, the table
-  // lookups either zero or leave unchanged the characters exceeding 0xF whereas
-  // on x64, the equivalent instruction (pshufb) automatically applies a mask,
-  // ignoring the 4 most significant bits. Thus the x64 implementation is
-  // optimized differently. This being said, if you use this code strictly
-  // just for minification (or just to identify the structural characters),
-  // there is a small untaken optimization opportunity here. We deliberately
-  // do not pick it up.
+  const uint8x16_t match_op_0 = vceqq_u8(vqtbl1q_u8(op_table, vshrq_n_u8(vaddq_u8(d0_0, vdupq_n_u8(3)), 4)), d0_0);
+  const uint8x16_t match_op_1 = vceqq_u8(vqtbl1q_u8(op_table, vshrq_n_u8(vaddq_u8(d0_1, vdupq_n_u8(3)), 4)), d0_1);
+  const uint8x16_t match_op_2 = vceqq_u8(vqtbl1q_u8(op_table, vshrq_n_u8(vaddq_u8(d0_2, vdupq_n_u8(3)), 4)), d0_2);
+  const uint8x16_t match_op_3 = vceqq_u8(vqtbl1q_u8(op_table, vshrq_n_u8(vaddq_u8(d0_3, vdupq_n_u8(3)), 4)), d0_3);

-  uint64_t op = simd8x64<bool>(
-        v.chunks[0].any_bits_set(0x7),
-        v.chunks[1].any_bits_set(0x7),
-        v.chunks[2].any_bits_set(0x7),
-        v.chunks[3].any_bits_set(0x7)
-  ).to_bitmask();
+  const uint8x16_t match_ws_0 = vqtbx1q_u8(vceqq_u8(d0_0, vdupq_n_u8(' ')), ws_table, d0_0);
+  const uint8x16_t match_ws_1 = vqtbx1q_u8(vceqq_u8(d0_1, vdupq_n_u8(' ')), ws_table, d0_1);
+  const uint8x16_t match_ws_2 = vqtbx1q_u8(vceqq_u8(d0_2, vdupq_n_u8(' ')), ws_table, d0_2);
+  const uint8x16_t match_ws_3 = vqtbx1q_u8(vceqq_u8(d0_3, vdupq_n_u8(' ')), ws_table, d0_3);

-  uint64_t whitespace = simd8x64<bool>(
-        v.chunks[0].any_bits_set(0x18),
-        v.chunks[1].any_bits_set(0x18),
-        v.chunks[2].any_bits_set(0x18),
-        v.chunks[3].any_bits_set(0x18)
-  ).to_bitmask();
+  const uint8x16_t bit_mask = simd8<uint8_t>(
+    0x01, 0x02, 0x04, 0x08, 0x10, 0x20, 0x40, 0x80,
+    0x01, 0x02, 0x04, 0x08, 0x10, 0x20, 0x40, 0x80
+  );
+
+  uint8x16_t op_sum0 = vpaddq_u8(vandq_u8(match_op_0, bit_mask), vandq_u8(match_op_1, bit_mask));
+  uint8x16_t ws_sum0 = vpaddq_u8(vandq_u8(match_ws_0, bit_mask), vandq_u8(match_ws_1, bit_mask));
+  uint8x16_t op_sum1 = vpaddq_u8(vandq_u8(match_op_2, bit_mask), vandq_u8(match_op_3, bit_mask));
+  uint8x16_t ws_sum1 = vpaddq_u8(vandq_u8(match_ws_2, bit_mask), vandq_u8(match_ws_3, bit_mask));
+  op_sum0 = vpaddq_u8(op_sum0, op_sum1);
+  ws_sum0 = vpaddq_u8(ws_sum0, ws_sum1);
+  op_sum0 = vpaddq_u8(op_sum0, op_sum0);
+  ws_sum0 = vpaddq_u8(ws_sum0, ws_sum0);
+  const uint64_t op = vgetq_lane_u64(vreinterpretq_u64_u8(op_sum0), 0);
+  const uint64_t whitespace = vgetq_lane_u64(vreinterpretq_u64_u8(ws_sum0), 0);

   return { whitespace, op };
 }
@@ -15498,7 +21893,7 @@ simdjson_warn_unused error_code implementation::minify(const uint8_t *buf, size_
   return arm64::stage1::json_minifier::minify<64>(buf, len, dst, dst_len);
 }

-simdjson_warn_unused error_code dom_parser_implementation::stage1(const uint8_t *_buf, size_t _len, stage1_mode streaming) noexcept {
+simdjson_flatten simdjson_warn_unused error_code dom_parser_implementation::stage1(const uint8_t *_buf, size_t _len, stage1_mode streaming) noexcept {
   this->buf = _buf;
   this->len = _len;
   return arm64::stage1::json_structural_indexer::index<64>(buf, len, *this, streaming);
@@ -15721,16 +22116,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
 }
 #endif

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
-                                uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
-  return _addcarry_u64(0, value1, value2,
-                       reinterpret_cast<unsigned __int64 *>(result));
-#else
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-#endif
-}

 } // unnamed namespace
 } // namespace haswell
@@ -16483,6 +22868,9 @@ namespace atomparsing {
 // to the compile-time constant 1936482662.
 simdjson_inline uint32_t string_to_uint32(const char* str) { uint32_t val; std::memcpy(&val, str, sizeof(uint32_t)); return val; }

+// Acts on the same principle as string_to_uint32, but on an 8-byte block of memory
+simdjson_inline uint64_t string_to_uint64(const char* str) { uint64_t val; std::memcpy(&val, str, sizeof(uint64_t)); return val; }
+

 // Again in str4ncmp we use a memcpy to avoid undefined behavior. The memcpy may appear expensive.
 // Yet all decent optimizing compilers will compile memcpy to a single instruction, just about.
@@ -16494,6 +22882,28 @@ simdjson_inline uint32_t str4ncmp(const uint8_t *src, const char* atom) {
   return srcval ^ string_to_uint32(atom);
 }

+// Checks that the first 8 characters of the input string match the given atom in a case-insensitive manner.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint64_t str8ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+  uint64_t srcval; // we want to avoid unaligned 32-bit loads (undefined in C/C++)
+  static_assert(sizeof(uint64_t) <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be larger than 8 bytes");
+  std::memcpy(&srcval, src, sizeof(uint64_t));
+
+  return (srcval | 0x2020202020202020ull) ^ string_to_uint64(atom);
+}
+
+// Checks that the first 3 characters of 'src' match 'atom' in a case-insensitive way.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint32_t str3ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+  return ((src[0] | 0x20) ^ atom[0]) //
+       | ((src[1] | 0x20) ^ atom[1]) //
+       | ((src[2] | 0x20) ^ atom[2]);
+}
+
 simdjson_warn_unused
 simdjson_inline bool is_valid_true_atom(const uint8_t *src) {
   return (str4ncmp(src, "true") | jsoncharutils::is_not_structural_or_whitespace(src[4])) == 0;
@@ -16530,6 +22940,71 @@ simdjson_inline bool is_valid_null_atom(const uint8_t *src, size_t len) {
   else { return false; }
 }

+#if SIMDJSON_ENABLE_NAN_INF
+// "nan" is 3 bytes; we check characters and then verify the next
+// character is structural or whitespace. We accept both "nan" and "NaN".
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src) {
+  return (str3ncmp_case_insensitive(src, "nan")
+        | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+}
+
+// checks that the next four characters of a string are 'nan"', where the 'nan'
+// is checked in a case-insensitive way.
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_in_string(const uint8_t *src) {
+  return (str3ncmp_case_insensitive(src, "nan") | (src[3] ^ '"')) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src, size_t len) {
+  if (len > 3) { return is_valid_nan_atom(src); }
+  if (len == 3) { return str3ncmp_case_insensitive(src, "nan") == 0; }
+  return false;
+}
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src) {
+  bool is_short_inf = (str3ncmp_case_insensitive(src, "inf")
+                     | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+  if(is_short_inf) return true;
+
+  // Check for 'infinity' (any capitalization)
+  return (str8ncmp_case_insensitive(src, "infinity") | jsoncharutils::is_not_structural_or_whitespace(src[8])) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_in_string(const uint8_t *src) {
+  bool is_short_inf = (str3ncmp_case_insensitive(src, "inf") | (src[3] ^ '"')) == 0;
+  if(is_short_inf) return true;
+
+  return (str8ncmp_case_insensitive(src, "infinity") | (src[8] ^ '"')) == 0;
+}
+
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src, size_t len) {
+  if (len > 8) { return is_valid_inf_atom(src); }
+  if (len == 8 && str8ncmp_case_insensitive(src, "infinity") == 0) {
+    return true;
+  }
+  if (len > 3) {
+    return (str3ncmp_case_insensitive(src, "inf")
+          | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+  }
+  if (len == 3) { return str3ncmp_case_insensitive(src, "inf") == 0; }
+  return false;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
 } // namespace atomparsing
 } // unnamed namespace
 } // namespace haswell
@@ -16620,7 +23095,7 @@ inline simdjson_warn_unused error_code dom_parser_implementation::set_capacity(s
 }

 inline simdjson_warn_unused error_code dom_parser_implementation::set_max_depth(size_t max_depth) noexcept {
-  if(max_depth == 0 || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
+  if(max_depth == 0 || max_depth > SIMDJSON_MAX_DEPTH || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
   // Stage 2 stacks
   open_containers.reset(new (std::nothrow) open_container[max_depth]);
   is_array.reset(new (std::nothrow) bool[max_depth]);
@@ -16799,6 +23274,7 @@ protected:
 /* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_NUMBERPARSING_H */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/base.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/jsoncharutils.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/atomparsing.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/internal/numberparsing_tables.h" */
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */

@@ -16838,6 +23314,18 @@ simdjson_inline double to_double(uint64_t mantissa, uint64_t real_exponent, bool
     return d;
 }

+// Convert a mantissa, an exponent and a sign bit into an ieee32 float (binary32).
+// The real_exponent needs to be in [0, 254] (technically real_exponent = 255 would be acceptable).
+// The mantissa should be in [0,1<<24). The bit at index (1U << 23) will be zeroed.
+simdjson_inline float to_float(uint32_t mantissa, uint32_t real_exponent, bool negative) {
+    float f;
+    mantissa &= ~(uint32_t(1) << 23);
+    mantissa |= real_exponent << 23;
+    mantissa |= ((static_cast<uint32_t>(negative)) << 31);
+    std::memcpy(&f, &mantissa, sizeof(f));
+    return f;
+}
+
 // Attempts to compute i * 10^(power) exactly; and if "negative" is
 // true, negate the result.
 // This function will only work in some cases, when it does not work, success is
@@ -17094,6 +23582,219 @@ simdjson_inline bool compute_float_64(int64_t power, uint64_t i, bool negative,
   return true;
 }

+// Attempts to compute i * 10^(power) as a binary32 (float) value; and if
+// "negative" is true, negate the result.
+//
+// This is the same Eisel-Lemire algorithm as compute_float_64, retargeted at
+// binary32. It is adapted from fast_float
+// (https://github.com/fastfloat/fast_float), where the routine is written once
+// and instantiated for each binary format. We only need 24 bits of mantissa
+// instead of 53, so we ask the truncated product for 24 + 2 = 26 accurate bits
+// (plus one bit that may be lost to the "upperbit" shift), and we shift the
+// product right by (upperbit + 64 - 23 - 3) instead of (upperbit + 64 - 52 - 3).
+//
+// The power-of-five table is shared with the binary64 code: the useful range
+// for binary32, [smallest_power_binary32, largest_power_binary32], is a strict
+// subset of [smallest_power, largest_power].
+//
+// The function returns false when the result would be infinite: simdjson
+// refuses to parse infinite values, so the caller reports an error. Unlike
+// compute_float_64, a false return never means "try harder": the accuracy of
+// the two-limb product is guaranteed by Noble Mushtak and Daniel Lemire,
+// "Fast Number Parsing Without Fallback" (https://arxiv.org/abs/2212.06644).
+// The caller must still fall back to a slow path when the 64-bit mantissa i
+// was truncated (more than 19 significant digits).
+//
+// We assume that power is in the [smallest_power, largest_power] interval: the
+// caller is responsible for this check.
+simdjson_inline bool compute_float_32(int64_t power, uint64_t i, bool negative, float &d) {
+  // Powers of ten that are exactly representable as binary32 values: 10^k is
+  // exact as long as k <= 10 (5^10 = 9765625 < 2**24 while 5^11 > 2**24).
+  static constexpr float power_of_ten_float[] = {
+      1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+      1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+  // The range of powers of ten that a non-zero, finite binary32 value can be
+  // built from. Anything smaller rounds to zero, anything larger is infinite.
+  // These are the smallest_power_of_ten()/largest_power_of_ten() constants that
+  // fast_float uses for binary32.
+  constexpr int smallest_power_binary32 = -65;
+  constexpr int largest_power_binary32 = 38;
+
+  // We start with the fast path described in
+  // Clinger WD. How to read floating point numbers accurately.
+  // ACM SIGPLAN Notices. 1990
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+  // We cannot be certain that x/y is rounded to nearest.
+  if (0 <= power && power <= 10 && i <= 16777215)
+#else
+  if (-10 <= power && power <= 10 && i <= 16777215)
+#endif
+  {
+    // Convert the integer into a float. This is lossless since
+    // 0 <= i <= 2^24 - 1.
+    d = float(i);
+    // Both d and the power of ten are exactly representable as binary32
+    // values, so the product (or quotient) is correctly rounded.
+    if (power < 0) {
+      d = d / power_of_ten_float[-power];
+    } else {
+      d = d * power_of_ten_float[power];
+    }
+    if (negative) {
+      d = -d;
+    }
+    return true;
+  }
+
+  // The fast path has failed, so we fall back on the Eisel-Lemire algorithm.
+  // It needs i > 0 (so that the leading bit of i can be normalized), so we
+  // handle i == 0 separately. We also handle the powers of ten that are so
+  // small (or so large) that the answer is zero (or infinite) whatever the
+  // mantissa is.
+  if (i == 0 || power < smallest_power_binary32) {
+    d = negative ? -0.0f : 0.0f;
+    return true;
+  }
+  if (power > largest_power_binary32) {
+    // We have, for sure, an infinite value.
+    return false;
+  }
+
+  // The exponent is 128 + 63 + power + floor(log(5**power)/log(2)).
+  // The 128 comes from the ieee32 standard (the minimal exponent is -127).
+  // The 63 comes from the fact that we use a 64-bit word.
+  // See compute_float_64 for a discussion of the magical 152170 + 65536.
+  int64_t exponent = (((152170 + 65536) * power) >> 16) + 128 + 63;
+
+  // We want the most significant bit of i to be 1. Shift if needed.
+  int lz = leading_zeroes(i);
+  i <<= lz;
+
+  // We want the most significant 64 bits of the product i * 5**power. It is
+  // safe to index the table because
+  // smallest_power <= smallest_power_binary32 <= power
+  //               <= largest_power_binary32 <= largest_power.
+  const uint32_t index = 2 * uint32_t(power - simdjson::internal::smallest_power);
+#if SIMDJSON_STATIC_REFLECTION
+  simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index]);
+#else
+  simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index]);
+#endif
+
+  // Unless the least significant 38 bits of the high (64-bit) part of the full
+  // product are all 1s, then we know that the most significant 26 bits are
+  // exact and no further work is needed. Having 26 bits is necessary because
+  // we need 24 bits for the mantissa but we have to have one rounding bit and
+  // we can waste a bit if the most significant bit of the product is zero.
+  // (For binary64, the same reasoning gives 55 bits and a 9-bit mask.)
+  if((firstproduct.high & 0x3FFFFFFFFF) == 0x3FFFFFFFFF) {
+    // The truncated multiplication was not accurate enough; use the next 64
+    // bits of the power of five to refine it. See compute_float_64 for a
+    // detailed discussion.
+#if SIMDJSON_STATIC_REFLECTION
+    simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index + 1]);
+#else
+    simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index + 1]);
+#endif
+    firstproduct.low += secondproduct.high;
+    if(secondproduct.high > firstproduct.low) { firstproduct.high++; }
+  }
+  uint64_t lower = firstproduct.low;
+  uint64_t upper = firstproduct.high;
+  // The final mantissa should be 24 bits with a leading 1.
+  // We shift it so that it occupies 25 bits with a leading 1.
+  ///////
+  uint64_t upperbit = upper >> 63;
+  uint64_t mantissa = upper >> (upperbit + 38); // 38 == 64 - 23 - 3
+  lz += int(1 ^ upperbit);
+
+  // Here we have mantissa < (1<<25).
+  int64_t real_exponent = exponent - lz;
+  if (simdjson_unlikely(real_exponent <= 0)) { // we have a subnormal?
+    // Here we have that real_exponent <= 0 so -real_exponent >= 0
+    if(-real_exponent + 1 >= 64) { // if we have more than 64 bits below the minimum exponent, you have a zero for sure.
+      d = negative ? -0.0f : 0.0f;
+      return true;
+    }
+    // next line is safe because -real_exponent + 1 < 64
+    mantissa >>= -real_exponent + 1;
+    // Thankfully, we can't have both "round-to-even" and subnormals because
+    // "round-to-even" only occurs for powers close to 0.
+    mantissa += (mantissa & 1); // round up
+    mantissa >>= 1;
+    // As in compute_float_64, rounding up may take us out of the subnormal
+    // range, so we can only decide after rounding.
+    real_exponent = (mantissa < (uint64_t(1) << 23)) ? 0 : 1;
+    d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+    return true;
+  }
+  // We have to round to even. The "to even" part is only a problem when we are
+  // right in between two floats, which we guard against. The bounds on the
+  // power of ten are those of fast_float's
+  // min_exponent_round_to_even()/max_exponent_round_to_even() for binary32:
+  // when q >= 0, (2m+1) must be divisible by 5^q with 5^q <= 2^25, so q <= 10;
+  // when q < 0, we need 2^24 x 5^{-q} < 2^{64}, so q >= -17.
+  if (simdjson_unlikely((lower <= 1) && (power >= -17) && (power <= 10) && ((mantissa & 3) == 1))) {
+    if((mantissa << (upperbit + 38)) == upper) {
+      mantissa &= ~uint64_t(1);   // flip it so that we do not round up
+    }
+  }
+
+  mantissa += mantissa & 1;
+  mantissa >>= 1;
+
+  // Here we have mantissa < (1<<24), unless there was an overflow
+  if (mantissa >= (uint64_t(1) << 24)) {
+    mantissa = (uint64_t(1) << 23);
+    real_exponent++;
+  }
+  mantissa &= ~(uint64_t(1) << 23);
+  // we have to check that real_exponent is in range, otherwise we bail out
+  if (simdjson_unlikely(real_exponent > 254)) {
+    // We have an infinite value!!! We could actually throw an error here if we could.
+    return false;
+  }
+  d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+  return true;
+}
+
+#if SIMDJSON_ENABLE_NAN_INF
+// Parses a nan or infinity. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, double& d) noexcept {
+  if (atomparsing::is_valid_inf_atom(src)) {
+    double inf = std::numeric_limits<double>::infinity();
+    d = negative ? -inf : inf;
+    return true;
+  }
+
+  if (atomparsing::is_valid_nan_atom(src)) {
+    d = std::numeric_limits<double>::quiet_NaN();
+    return true;
+  }
+
+  return false;
+}
+
+// Parses a nan or infinity as a binary32 value. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, float& d) noexcept {
+  if (atomparsing::is_valid_inf_atom(src)) {
+    float inf = std::numeric_limits<float>::infinity();
+    d = negative ? -inf : inf;
+    return true;
+  }
+
+  if (atomparsing::is_valid_nan_atom(src)) {
+    d = std::numeric_limits<float>::quiet_NaN();
+    return true;
+  }
+
+  return false;
+}
+#endif
+
 // We call a fallback floating-point parser that might be slow. Note
 // it will accept JSON numbers, but the JSON spec. is more restrictive so
 // before you call parse_float_fallback, you need to have validated the input
@@ -17131,6 +23832,16 @@ static bool parse_float_fallback(const uint8_t *ptr, const uint8_t *end_ptr, dou
   return !(*outDouble > (std::numeric_limits<double>::max)() || *outDouble < std::numeric_limits<double>::lowest());
 }

+// Same as parse_float_fallback, but for binary32 (float) values. Going through
+// the binary64 fallback and then rounding to binary32 would be subject to
+// double rounding, so we run the fallback algorithm directly on binary32.
+static bool parse_float_fallback(const uint8_t *ptr, float *outFloat) {
+  *outFloat = simdjson::internal::from_chars_float(reinterpret_cast<const char *>(ptr));
+  // We do not accept infinite values. See the binary64 version above for why we
+  // do not use std::isfinite.
+  return !(*outFloat > (std::numeric_limits<float>::max)() || *outFloat < std::numeric_limits<float>::lowest());
+}
+
 // check quickly whether the next 8 chars are made of digits
 // at a glance, it looks better than Mula's
 // http://0x80.pl/articles/swar-digits-validate.html
@@ -17149,6 +23860,28 @@ simdjson_inline bool is_made_of_eight_digits_fast(const uint8_t *chars) {
           0x3333333333333333);
 }

+// The same idea for four characters. A block of eight only bites when eight
+// digits are left, so a run of 4 to 7 digits used to fall back to one digit at a
+// time; taking four of them at once is what makes the tail of a long fraction
+// cheap. Adapted from fast_float, credit @aqrit.
+simdjson_inline bool is_made_of_four_digits_fast(const uint8_t *chars) {
+  uint32_t val;
+  // Reads up to 3 bytes beyond the digits, which SIMDJSON_PADDING covers.
+  static_assert(3 <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be bigger than 3");
+  std::memcpy(&val, chars, 4);
+  return !(((val + 0x46464646) | (val - 0x30303030)) & 0x80808080);
+}
+
+// Only call this when is_made_of_four_digits_fast() says the four characters are
+// digits. Little-endian: guarded by SIMDJSON_SWAR_NUMBER_PARSING.
+simdjson_inline uint32_t parse_four_digits_unrolled(const uint8_t *chars) {
+  uint32_t val;
+  std::memcpy(&val, chars, 4);
+  val -= 0x30303030;
+  val = (val * 10) + (val >> 8);
+  return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
 template<typename I>
 SIMDJSON_NO_SANITIZE_UNDEFINED // We deliberately allow overflow here and check later
 simdjson_inline bool parse_digit(const uint8_t c, I &i) {
@@ -17165,26 +23898,73 @@ simdjson_inline bool is_digit(const uint8_t c) {
   return static_cast<uint8_t>(c - '0') <= 9;
 }

-simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
-  // we continue with the fiction that we have an integer. If the
-  // floating point number is representable as x * 10^z for some integer
-  // z that fits in 53 bits, then we will be able to convert back the
-  // the integer into a float in a lossless manner.
-  const uint8_t *const first_after_period = p;
+// Consumes a run of digits into i, eight at a time while eight are available,
+// then four, then one at a time. Overflow is deliberate: the caller counts the
+// digits and falls back when there are too many for a 64-bit mantissa.
+//
+// Long runs of digits are what a float-heavy document is made of: the fraction
+// of a binary64 printed to full precision is 15 to 17 digits, and reading those
+// one at a time is the single largest cost in parse_double(). Integer parts are
+// left alone: they are usually a handful of digits, and there the check for a
+// block of eight is wasted work.
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_fraction_digits(const uint8_t *&p, uint64_t &i) {
+#ifdef SIMDJSON_SWAR_NUMBER_PARSING
+#if SIMDJSON_SWAR_NUMBER_PARSING
+  while (is_made_of_eight_digits_fast(p)) {
+    i = i * 100000000 + parse_eight_digits_unrolled(p);
+    p += 8;
+  }
+  // A 4 to 7 digit remainder is the common case once the blocks of eight are
+  // gone: 15 fraction digits leave 7, and 17 leave 1 after two blocks.
+  if (is_made_of_four_digits_fast(p)) {
+    i = i * 10000 + parse_four_digits_unrolled(p);
+    p += 4;
+  }
+#endif // SIMDJSON_SWAR_NUMBER_PARSING
+#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
+  while (parse_digit(*p, i)) { p++; }
+}
+
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_float_integer_digits(const uint8_t *&p, uint64_t &i,
+                                                bool &leading_zero) {
+  if (!parse_digit(*p, i)) { leading_zero = true; return; }
+  p++;
+  leading_zero = (i == 0);
+  while (parse_digit(*p, i)) { p++; }
+}

+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_integer_digits(const uint8_t *&p, uint64_t &i) {
 #ifdef SIMDJSON_SWAR_NUMBER_PARSING
 #if SIMDJSON_SWAR_NUMBER_PARSING
-  // this helps if we have lots of decimals!
-  // this turns out to be frequent enough.
+  // Identifiers, timestamps and counters often have eight digits or more.
+  // Prior related work: jsonifier parses integers as eight-digit SWAR words
+  // (str_to_i.hpp, https://github.com/nihilai-collective/Jsonifier). This
+  // takes one such word with parse_eight_digits_unrolled, the routine
+  // simdjson uses for long fractions.
   if (is_made_of_eight_digits_fast(p)) {
     i = i * 100000000 + parse_eight_digits_unrolled(p);
     p += 8;
   }
+  const uint8_t *const swar_end = p + 8;
+  while (p < swar_end && is_made_of_four_digits_fast(p)) {
+    i = i * 10000 + parse_four_digits_unrolled(p);
+    p += 4;
+  }
 #endif // SIMDJSON_SWAR_NUMBER_PARSING
 #endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
-  // Unrolling the first digit makes a small difference on some implementations (e.g. westmere)
-  if (parse_digit(*p, i)) { ++p; }
   while (parse_digit(*p, i)) { p++; }
+}
+
+simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
+  // we continue with the fiction that we have an integer. If the
+  // floating point number is representable as x * 10^z for some integer
+  // z that fits in 53 bits, then we will be able to convert back the
+  // the integer into a float in a lossless manner.
+  const uint8_t *const first_after_period = p;
+  parse_fraction_digits(p, i);
   exponent = first_after_period - p;
   // Decimal without digits (123.) is illegal
   if (exponent == 0) {
@@ -17273,7 +24053,7 @@ simdjson_inline size_t significant_digits(const uint8_t * start_digits, size_t d
 } // unnamed namespace

 /** @private */
-static error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
+inline error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
   if (parse_float_fallback(src, answer)) {
     return SUCCESS;
   }
@@ -17356,15 +24136,17 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
   return SUCCESS;              // always succeeds
 }

-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept  { return false; }
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept  { return false; }
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept { return number_type::signed_integer; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline bool is_negative(const uint8_t *) noexcept  { return false; }
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t *) noexcept  { return false; }
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t *) noexcept { return number_type::signed_integer; }
 #else

 // parse the number at src
@@ -17395,7 +24177,21 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
   size_t digit_count = size_t(p - start_digits);
-  if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) { return INVALID_NUMBER(src); }
+  if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // By this point, we know that our input does not begin with a digit. We will attempt
+    // to handle NaN/Infinity.
+
+    double d;
+    if (compute_nan_inf(p, negative, d)) {
+      writer.append_double(d);
+      return SUCCESS;
+    }
+#endif
+
+    return INVALID_NUMBER(src);
+  }

   //
   // Handle floats if there is a . or e (or both)
@@ -17444,7 +24240,7 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
     // - Therefore, if the number is positive and lower than that, it's overflow.
     // - The value we are looking at is less than or equal to INT64_MAX.
     //
-    }  else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return INVALID_NUMBER(src); }
+    }  else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return BIGINT_NUMBER(src); }
   }

   // Write unsigned if it does not fit in a signed integer.
@@ -17543,7 +24339,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const u
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -17641,7 +24437,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_stri
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -17696,7 +24492,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uin
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -17782,7 +24578,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = src;
   uint64_t i = 0;
-  while (parse_digit(*src, i)) { src++; }
+  parse_integer_digits(src, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -17822,11 +24618,20 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   //
   uint64_t i = 0;
   const uint8_t *p = src;
-  p += parse_digit(*p, i);
-  bool leading_zero = (i == 0);
-  while (parse_digit(*p, i)) { p++; }
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
   // no integer digits, or 0123 (zero must be solo)
-  if ( p == src ) { return INCORRECT_TYPE; }
+  if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no loading digits, the number may be nan or infinity.
+    // Attempt to compute those, and return on success.
+    double d;
+    if (compute_nan_inf(p, negative, d)) { return d; }
+#endif
+
+    return INCORRECT_TYPE;
+  }
   if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }

   //
@@ -17837,9 +24642,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   if (simdjson_likely(*p == '.')) {
     p++;
     const uint8_t *start_decimal_digits = p;
-    if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
-    p++;
-    while (parse_digit(*p, i)) { p++; }
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
     exponent = -(p - start_decimal_digits);

     // Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -17888,6 +24692,97 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   return d;
 }

+// Parse a JSON number into a binary32 (float) value.
+//
+// This mirrors parse_double, but it rounds to binary32 directly instead of
+// rounding to binary64 and then to binary32: the latter is subject to double
+// rounding and would not always produce the nearest float.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * src) noexcept {
+  //
+  // Check for minus sign
+  //
+  bool negative = (*src == '-');
+  src += uint8_t(negative);
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, the number may be nan or infinity.
+    // Attempt to compute those, and return on success.
+    float f;
+    if (compute_nan_inf(p, negative, f)) { return f; }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = p-start_digits > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if (*p == 'e' || *p == 'E') {
+    p++;
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while (parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  float d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_32(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
 simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
   return (*src == '-');
 }
@@ -18040,11 +24935,25 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   //
   uint64_t i = 0;
   const uint8_t *p = src;
-  p += parse_digit(*p, i);
-  bool leading_zero = (i == 0);
-  while (parse_digit(*p, i)) { p++; }
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
   // no integer digits, or 0123 (zero must be solo)
-  if ( p == src ) { return INCORRECT_TYPE; }
+  if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, attempt to parse numbers that are either
+    // NaN or Infinity
+    if (atomparsing::is_valid_inf_in_string(src)) {
+      double inf = std::numeric_limits<double>::infinity();
+      return negative ? -inf : inf;
+    }
+
+    if (atomparsing::is_valid_nan_in_string(src)) {
+      return std::numeric_limits<double>::quiet_NaN();
+    }
+#endif
+
+    return INCORRECT_TYPE;
+  }
   if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }

   //
@@ -18055,9 +24964,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   if (simdjson_likely(*p == '.')) {
     p++;
     const uint8_t *start_decimal_digits = p;
-    if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
-    p++;
-    while (parse_digit(*p, i)) { p++; }
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
     exponent = -(p - start_decimal_digits);

     // Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -18106,6 +25014,99 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   return d;
 }

+// Parse a JSON number held inside a JSON string into a binary32 (float) value.
+// See parse_float for why we do not simply round parse_double_in_string.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * src) noexcept {
+  //
+  // Check for minus sign
+  //
+  bool negative = (*(src + 1) == '-');
+  src += uint8_t(negative) + 1;
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, attempt to parse numbers that are either
+    // NaN or Infinity
+    if (atomparsing::is_valid_inf_in_string(src)) {
+      float inf = std::numeric_limits<float>::infinity();
+      return negative ? -inf : inf;
+    }
+
+    if (atomparsing::is_valid_nan_in_string(src)) {
+      return std::numeric_limits<float>::quiet_NaN();
+    }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = p-start_digits > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if (*p == 'e' || *p == 'E') {
+    p++;
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while (parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if (*p != '"') { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  float d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_32(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
 } // unnamed namespace
 #endif // SIMDJSON_SKIPNUMBERPARSING

@@ -18465,16 +25466,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
 }
 #endif

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
-                                uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
-  return _addcarry_u64(0, value1, value2,
-                       reinterpret_cast<unsigned __int64 *>(result));
-#else
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-#endif
-}

 } // unnamed namespace
 } // namespace haswell
@@ -20024,6 +27015,296 @@ simdjson_inline uint32_t find_next_document_index(dom_parser_implementation &par
   return 0;
 }

+/**
+ * Sentinel value returned to indicate a document started but didn't fit
+ * (CAPACITY error), as opposed to 0 which means no document content found
+ * (EMPTY).
+ */
+constexpr uint32_t DOCUMENT_TOO_LARGE = UINT32_MAX;
+
+/**
+ * For RFC 7464 JSON text sequences, filter RS from structural indexes and
+ * find batch boundaries.
+ *
+ * In JSON sequence mode, RS (0x1E) marks the start of each JSON text.
+ * RS bytes appear in structural_indexes as they are classified as scalars.
+ * This function:
+ * 1. Scans structural_indexes to find and count RS positions
+ * 2. Filters RS out of structural_indexes in-place
+ * 3. Determines batch boundaries based on RS positions
+ *
+ * @param parser The parser with structural_indexes and buf.
+ * @param len The length of the current batch buffer.
+ * @param is_final True if this is the final batch (no more data coming).
+ * @param next_batch_start Output: offset where the next batch should start.
+ * @param scan_len Offset past which no value start may be derived. Equals len
+ *        unless stage 1 dropped a trailing unclosed string, whose bytes it
+ *        never validated.
+ * @return The number of structural indexes to keep (after RS filtering),
+ *         0 if no document content found (EMPTY),
+ *         or DOCUMENT_TOO_LARGE if a document started but didn't fit (CAPACITY).
+ */
+simdjson_inline uint32_t find_next_document_index_json_sequence(
+    dom_parser_implementation &parser,
+    size_t len,
+    bool is_final,
+    uint32_t &next_batch_start,
+    size_t scan_len) {
+  // Default: next batch starts at end of buffer
+  next_batch_start = uint32_t(len);
+
+  if (parser.n_structural_indexes == 0) { return 0; }
+
+  // Phase 1: Scan structural_indexes to find RS positions and handle them.
+  // RS marks the start of a JSON text. For objects/arrays, the '{' or '[' after RS
+  // is already in structural_indexes (it's an operator). For scalars like numbers,
+  // the digit following RS is NOT in structural_indexes because the scanner sees
+  // RS as a scalar, making the digit a scalar continuation, not a start.
+  // We must: (1) remove RS from structural_indexes, and (2) for scalars, add the
+  // actual value start position.
+  // The EOF sentinel: len, or where a discarded unclosed string starts.
+  const uint32_t sentinel = parser.structural_indexes[parser.n_structural_indexes];
+  uint32_t write_idx = 0;
+  uint32_t last_rs_pos = 0;
+  uint32_t rs_count = 0;
+
+  for (uint32_t read_idx = 0; read_idx < parser.n_structural_indexes; read_idx++) {
+    const uint32_t pos = parser.structural_indexes[read_idx];
+    if (parser.buf[pos] == 0x1E) {
+      // This is an RS character - find the actual JSON value start.
+      last_rs_pos = pos;
+      rs_count++;
+      // Skip past this RS and any whitespace *and any additional RSes*
+      // to locate the real value. Consecutive RSes are degenerate
+      // "empty records" per RFC 7464; we collapse them here. They do
+      // not always appear as separate entries in structural_indexes
+      // because the scanner groups runs of adjacent non-whitespace
+      // scalar bytes (including RS) into a single scalar start.
+      uint32_t value_start = pos + 1;
+      while (value_start < scan_len) {
+        const uint8_t c = parser.buf[value_start];
+        if (c == ' ' || c == '\t' || c == '\n' || c == '\r') {
+          value_start++;
+        } else if (c == 0x1E) {
+          // Collapsed empty record. Still count it so rs_count reflects
+          // the true number of record markers and last_rs_pos tracks
+          // the final one.
+          last_rs_pos = value_start;
+          rs_count++;
+          value_start++;
+        } else {
+          break;
+        }
+      }
+      // If the scanner emitted additional structurals inside the
+      // whitespace+RS run we just walked over (i.e., isolated RSes
+      // separated by whitespace), skip past them so we do not
+      // double-count or double-emit.
+      while (read_idx + 1 < parser.n_structural_indexes &&
+             parser.structural_indexes[read_idx + 1] < value_start) {
+        read_idx++;
+      }
+      // Check if the value start is an operator (always present in
+      // scanner structural_indexes) or a scalar-like start (which may
+      // be missing from structural_indexes and must be added here).
+      // Note: '"' is NOT always in structural_indexes. The scanner
+      // classifies '"' as a scalar character and emits it as a
+      // structural only when it is a *scalar start* (preceded by
+      // whitespace or an operator). When '"' immediately follows an
+      // RS (which the scanner also classifies as scalar), it is
+      // treated as a scalar continuation and not emitted - so we
+      // must add it here just like any other scalar value.
+      if (value_start < scan_len) {
+        const uint8_t c = parser.buf[value_start];
+        const bool is_operator =
+            (c == '{' || c == '}' || c == '[' || c == ']' ||
+             c == ':' || c == ',');
+        // If the next scanner structural is exactly at value_start,
+        // the scanner already emitted it (it followed whitespace) and
+        // we must not add a duplicate - a subsequent iteration will
+        // copy it into write_idx.
+        const bool already_emitted =
+            (read_idx + 1 < parser.n_structural_indexes &&
+             parser.structural_indexes[read_idx + 1] == value_start);
+        if (!is_operator && !already_emitted) {
+          // Scalar value (number/true/false/null/string) - add its
+          // position since scanner missed it.
+          parser.structural_indexes[write_idx++] = value_start;
+        }
+      }
+    } else {
+      // Not RS, copy to output
+      parser.structural_indexes[write_idx++] = pos;
+    }
+  }
+
+  // Update structural index count. Compaction left a stale index in the slot
+  // past the end: restore the EOF sentinel that stage 1 had planted there, which
+  // document_stream::truncated_bytes() reads after a final batch.
+  parser.n_structural_indexes = write_idx;
+  parser.structural_indexes[write_idx] = sentinel;
+
+  if (parser.n_structural_indexes == 0) {
+    // Only RS markers here: the last one opens a record continuing past the
+    // window, so that is where the next batch resumes.
+    if (!is_final && rs_count > 0) { next_batch_start = last_rs_pos; }
+    return 0;
+  }
+  if (rs_count == 0) {
+    // No RS found; for final batch, try generic boundary detection
+    return is_final ? find_next_document_index(parser) : 0;
+  }
+
+  // Phase 2: Determine batch boundaries based on RS positions
+
+  if (is_final) {
+    // Final batch: all documents are complete (last one ends at EOF).
+    // In json_sequence mode, RS markers define document boundaries, so all
+    // remaining structurals form complete documents. Return them all directly.
+    // (Calling find_next_document_index() would fail for scalar documents.)
+    return parser.n_structural_indexes;
+  }
+
+  // Partial batch: need to find complete documents only.
+  // A document starting at an RS is complete if there is another RS after it.
+  next_batch_start = last_rs_pos;
+
+  if (rs_count < 2) {
+    // Only one RS, so we have at most one document that may be incomplete.
+    // We cannot confirm it is complete without another RS.
+    // Return DOCUMENT_TOO_LARGE if content was found (write_idx > 0), 0 if only separators.
+    return (parser.n_structural_indexes > 0) ? DOCUMENT_TOO_LARGE : 0;
+  }
+
+  // We have at least 2 RS markers. The last complete document ends before last_rs_pos.
+
+  // Find the structural index cutoff: keep only structurals < last_rs_pos.
+  // Since we already filtered RS, all remaining structurals are valid.
+  // We iterate backward to find the last structural before last_rs_pos.
+  uint32_t keep_count = 0;
+  for (uint32_t i = parser.n_structural_indexes; i > 0; i--) {
+    if (parser.structural_indexes[i - 1] < last_rs_pos) {
+      keep_count = i;
+      break;
+    }
+  }
+
+  // No structurals before the last RS - no complete documents
+  if (keep_count == 0) { return 0; }
+
+  // All documents before the last RS are complete by definition (the next RS
+  // confirms their end). No need to call find_next_document_index() which
+  // would fail for scalar documents like `1` or `"hello"`.
+  return keep_count;
+}
+
+/**
+ * Filter comma-delimited documents by removing root-level commas from
+ * structural indexes.
+ *
+ * For comma-delimited format like `{...},{...},{...}`, we need to remove
+ * the commas that separate documents (depth 0) while preserving commas
+ * inside arrays and objects (depth > 0).
+ *
+ * After filtering, the structural indexes look like whitespace-delimited
+ * documents, so find_next_document_index() works unchanged.
+ *
+ * @param parser The parser with structural_indexes and buf.
+ * @param len The length of the current batch buffer.
+ * @param is_final True if this is the final batch (no more data coming).
+ * @param next_batch_start Output: offset where the next batch should start.
+ * @return The number of structural indexes to keep,
+ *         0 if no document content found (EMPTY),
+ *         or DOCUMENT_TOO_LARGE if a document started but didn't fit (CAPACITY).
+ */
+simdjson_inline uint32_t filter_comma_delimited(
+    dom_parser_implementation &parser,
+    size_t len,
+    bool is_final,
+    uint32_t &next_batch_start) {
+  // Default: next batch starts at end of buffer
+  next_batch_start = uint32_t(len);
+
+  if (parser.n_structural_indexes == 0) { return 0; }
+
+  // Track depth to identify root-level commas (depth 0)
+  int depth = 0;
+  // The EOF sentinel: len, or where a discarded unclosed string starts.
+  const uint32_t sentinel = parser.structural_indexes[parser.n_structural_indexes];
+  uint32_t write_idx = 0;
+  uint32_t last_root_comma_pos = 0;
+  uint32_t root_comma_count = 0;
+
+  for (uint32_t i = 0; i < parser.n_structural_indexes; i++) {
+    uint32_t idx = parser.structural_indexes[i];
+    uint8_t c = parser.buf[idx];
+
+    switch (c) {
+      case '{': case '[':
+        depth++;
+        parser.structural_indexes[write_idx++] = idx;
+        break;
+      case '}': case ']':
+        depth--;
+        parser.structural_indexes[write_idx++] = idx;
+        break;
+      case ',':
+        if (depth == 0) {
+          // Root-level comma = document boundary, skip it
+          last_root_comma_pos = idx;
+          root_comma_count++;
+          continue;
+        }
+        parser.structural_indexes[write_idx++] = idx;
+        break;
+      default:
+        // Colons, scalars, etc.
+        parser.structural_indexes[write_idx++] = idx;
+        break;
+    }
+  }
+
+  // Update structural index count. Compaction left a stale index in the slot
+  // past the end: restore the EOF sentinel that stage 1 had planted there, which
+  // document_stream::truncated_bytes() reads after a final batch.
+  parser.n_structural_indexes = write_idx;
+  parser.structural_indexes[write_idx] = sentinel;
+
+  if (parser.n_structural_indexes == 0) { return 0; }
+
+  if (is_final) {
+    // Final batch: use standard boundary detection on filtered indexes
+    return find_next_document_index(parser);
+  }
+
+  // Partial batch: need to find complete documents only.
+  // A document ending with a root comma is complete.
+  if (root_comma_count == 0) {
+    // No root commas found; we cannot confirm any document is complete.
+    // The whole batch might be one incomplete document.
+    // Return DOCUMENT_TOO_LARGE if content was found (write_idx > 0), 0 if only commas.
+    return (parser.n_structural_indexes > 0) ? DOCUMENT_TOO_LARGE : 0;
+  }
+
+  // We have at least one root comma. Documents before the last comma are complete.
+  next_batch_start = last_root_comma_pos + 1;
+
+  // Find the structural index cutoff: keep only structurals < last_root_comma_pos
+  uint32_t keep_count = 0;
+  for (uint32_t i = parser.n_structural_indexes; i > 0; i--) {
+    if (parser.structural_indexes[i - 1] < last_root_comma_pos) {
+      keep_count = i;
+      break;
+    }
+  }
+
+  if (keep_count == 0) { return 0; }
+
+  // Use standard boundary detection on the complete portion
+  parser.n_structural_indexes = keep_count;
+  return find_next_document_index(parser);
+}
+
 } // namespace stage1
 } // unnamed namespace
 } // namespace haswell
@@ -20457,7 +27738,6 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
         return EMPTY;
       }
     }
-
     parser.n_structural_indexes = new_structural_indexes;
   } else if (partial == stage1_mode::streaming_final) {
     if(have_unclosed_string) { parser.n_structural_indexes--; }
@@ -20485,6 +27765,75 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
         // the trailing garbage.
         return EMPTY;
     }
+  } else if (partial == stage1_mode::json_sequence_partial) {
+    // RFC 7464: use RS positions for batch boundaries
+    // A discarded unclosed string also caps how far the filter may scan.
+    size_t scan_len = len;
+    if(have_unclosed_string) {
+      parser.n_structural_indexes--;
+      if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return CAPACITY; }
+      scan_len = parser.structural_indexes[parser.n_structural_indexes];
+    }
+    uint32_t next_batch_start = uint32_t(len);
+    auto new_structural_indexes = find_next_document_index_json_sequence(parser, len, false, next_batch_start, scan_len);
+    if (new_structural_indexes == DOCUMENT_TOO_LARGE) {
+      return CAPACITY;
+    }
+    if (new_structural_indexes == 0) {
+      // An EMPTY batch must still advance next_batch_start, or document_stream
+      // re-parses the same bytes forever. CAPACITY when it cannot.
+      if (next_batch_start == 0) { return CAPACITY; }
+      parser.n_structural_indexes = 0;
+      parser.structural_indexes[0] = next_batch_start;
+      return EMPTY;
+    }
+    parser.n_structural_indexes = new_structural_indexes;
+    parser.structural_indexes[parser.n_structural_indexes] = next_batch_start;
+  } else if (partial == stage1_mode::json_sequence_final) {
+    // RFC 7464: final batch, last document extends to EOF
+    // As above: a discarded unclosed string caps how far the filter may scan.
+    size_t scan_len = len;
+    if(have_unclosed_string) {
+      parser.n_structural_indexes--;
+      scan_len = parser.structural_indexes[parser.n_structural_indexes];
+    }
+    uint32_t next_batch_start = uint32_t(len);
+    parser.n_structural_indexes = find_next_document_index_json_sequence(parser, len, true, next_batch_start, scan_len);
+    // The filter compacted structural_indexes in place and restored the EOF
+    // sentinel past the compacted end, so the copy below is either the start
+    // of a truncated document or len, as in streaming_final.
+    parser.structural_indexes[parser.n_structural_indexes + 1] = parser.structural_indexes[parser.n_structural_indexes];
+    parser.structural_indexes[parser.n_structural_indexes] = uint32_t(len);
+    if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return EMPTY; }
+  } else if (partial == stage1_mode::comma_delimited_partial) {
+    // Comma-delimited: filter root-level commas, use comma positions for batch boundaries
+    if(have_unclosed_string) {
+      parser.n_structural_indexes--;
+      if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return CAPACITY; }
+    }
+    uint32_t next_batch_start = uint32_t(len);
+    auto new_structural_indexes = filter_comma_delimited(parser, len, false, next_batch_start);
+    if (new_structural_indexes == DOCUMENT_TOO_LARGE) {
+      return CAPACITY;
+    }
+    if (new_structural_indexes == 0) {
+      // An EMPTY batch must still advance next_batch_start, or document_stream
+      // re-parses the same bytes forever. CAPACITY when it cannot.
+      if (next_batch_start == 0) { return CAPACITY; }
+      parser.n_structural_indexes = 0;
+      parser.structural_indexes[0] = next_batch_start;
+      return EMPTY;
+    }
+    parser.n_structural_indexes = new_structural_indexes;
+    parser.structural_indexes[parser.n_structural_indexes] = next_batch_start;
+  } else if (partial == stage1_mode::comma_delimited_final) {
+    // Comma-delimited: final batch, last document extends to EOF
+    if(have_unclosed_string) { parser.n_structural_indexes--; }
+    uint32_t next_batch_start = uint32_t(len);
+    parser.n_structural_indexes = filter_comma_delimited(parser, len, true, next_batch_start);
+    parser.structural_indexes[parser.n_structural_indexes + 1] = parser.structural_indexes[parser.n_structural_indexes];
+    parser.structural_indexes[parser.n_structural_indexes] = uint32_t(len);
+    if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return EMPTY; }
   }
   checker.check_eof();
   return checker.errors();
@@ -20567,7 +27916,6 @@ namespace {
 namespace stage2 {

 class json_iterator;
-class structural_iterator;
 struct tape_builder;
 struct tape_writer;

@@ -20864,7 +28212,7 @@ public:
    *
    * - increment_count(iter) - each time a value is found in an array or object.
    */
-  template<bool STREAMING, typename V>
+  template<bool STREAMING, bool UNPADDED, typename V>
   simdjson_warn_unused simdjson_inline error_code walk_document(V &visitor) noexcept;

   /**
@@ -20881,6 +28229,7 @@ public:
    *
    * They may include invalid JSON as well (such as `1.2.3` or `ture`).
    */
+  template<bool UNPADDED>
   simdjson_inline const uint8_t *peek() const noexcept;
   /**
    * Advance to the next token.
@@ -20889,6 +28238,7 @@ public:
    *
    * They may include invalid JSON as well (such as `1.2.3` or `ture`).
    */
+  template<bool UNPADDED>
   simdjson_inline const uint8_t *advance() noexcept;
   /**
    * Get the remaining length of the document, from the start of the current token.
@@ -20937,7 +28287,7 @@ public:
   simdjson_warn_unused simdjson_inline error_code visit_primitive(V &visitor, const uint8_t *value) noexcept;
 };

-template<bool STREAMING, typename V>
+template<bool STREAMING, bool UNPADDED, typename V>
 simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &visitor) noexcept {
   logger::log_start();

@@ -20952,7 +28302,7 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &
   // Read first value
   //
   {
-    auto value = advance();
+    auto value = advance<UNPADDED>();

     // Make sure the outer object or array is closed before continuing; otherwise, there are ways we
     // could get into memory corruption. See https://github.com/simdjson/simdjson/issues/906
@@ -20964,8 +28314,8 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &
     }

     switch (*value) {
-      case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
-      case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+      case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+      case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
       default: SIMDJSON_TRY( visitor.visit_root_primitive(*this, value) ); break;
     }
   }
@@ -20982,29 +28332,29 @@ object_begin:
   SIMDJSON_TRY( visitor.visit_object_start(*this) );

   {
-    auto key = advance();
+    auto key = advance<UNPADDED>();
     if (*key != '"') { log_error("Object does not start with a key"); return TAPE_ERROR; }
     SIMDJSON_TRY( visitor.increment_count(*this) );
     SIMDJSON_TRY( visitor.visit_key(*this, key) );
   }

 object_field:
-  if (simdjson_unlikely( *advance() != ':' )) { log_error("Missing colon after key in object"); return TAPE_ERROR; }
+  if (simdjson_unlikely( *advance<UNPADDED>() != ':' )) { log_error("Missing colon after key in object"); return TAPE_ERROR; }
   {
-    auto value = advance();
+    auto value = advance<UNPADDED>();
     switch (*value) {
-      case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
-      case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+      case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+      case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
       default: SIMDJSON_TRY( visitor.visit_primitive(*this, value) ); break;
     }
   }

 object_continue:
-  switch (*advance()) {
+  switch (*advance<UNPADDED>()) {
     case ',':
       SIMDJSON_TRY( visitor.increment_count(*this) );
       {
-        auto key = advance();
+        auto key = advance<UNPADDED>();
         if (simdjson_unlikely( *key != '"' )) { log_error("Key string missing at beginning of field in object"); return TAPE_ERROR; }
         SIMDJSON_TRY( visitor.visit_key(*this, key) );
       }
@@ -21032,16 +28382,16 @@ array_begin:

 array_value:
   {
-    auto value = advance();
+    auto value = advance<UNPADDED>();
     switch (*value) {
-      case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
-      case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+      case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+      case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
       default: SIMDJSON_TRY( visitor.visit_primitive(*this, value) ); break;
     }
   }

 array_continue:
-  switch (*advance()) {
+  switch (*advance<UNPADDED>()) {
     case ',': SIMDJSON_TRY( visitor.increment_count(*this) ); goto array_value;
     case ']': log_end_value("array"); SIMDJSON_TRY( visitor.visit_array_end(*this) ); goto scope_end;
     default: log_error("Missing comma between array values"); return TAPE_ERROR;
@@ -21069,11 +28419,29 @@ simdjson_inline json_iterator::json_iterator(dom_parser_implementation &_dom_par
     dom_parser{_dom_parser} {
 }

+// Stage 1 leaves a sentinel index of value `len`; reading buf[len] requires
+// padding. For UNPADDED, return a dummy so walk_document can fail cleanly (#2815).
+template<bool UNPADDED>
 simdjson_inline const uint8_t *json_iterator::peek() const noexcept {
-  return &buf[*(next_structural)];
+  const uint32_t idx = *(next_structural);
+  SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+    if (simdjson_unlikely(idx >= dom_parser.len)) {
+      static constexpr uint8_t unpadded_eof_sentinel = 0;
+      return &unpadded_eof_sentinel;
+    }
+  }
+  return &buf[idx];
 }
+template<bool UNPADDED>
 simdjson_inline const uint8_t *json_iterator::advance() noexcept {
-  return &buf[*(next_structural++)];
+  const uint32_t idx = *(next_structural++);
+  SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+    if (simdjson_unlikely(idx >= dom_parser.len)) {
+      static constexpr uint8_t unpadded_eof_sentinel = 0;
+      return &unpadded_eof_sentinel;
+    }
+  }
+  return &buf[idx];
 }
 simdjson_inline size_t json_iterator::remaining_len() const noexcept {
   return dom_parser.len - *(next_structural-1);
@@ -21113,7 +28481,20 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::visit_root_primit
     case '"': return visitor.visit_root_string(*this, value);
     case 't': return visitor.visit_root_true_atom(*this, value);
     case 'f': return visitor.visit_root_false_atom(*this, value);
+#if SIMDJSON_ENABLE_NAN_INF
+    case 'n': {
+      auto err = visitor.visit_root_null_atom(*this, value);
+      if (err == SUCCESS) { return err; }
+      // propagate the error value returned by a bad 'null' atom if parsing 'nan' fails
+      return visitor.visit_root_nan_atom(*this, value, err);
+    }
+    // 'N' isn't a canonically recognized atom, so we return a TAPE_ERROR if failure occurs
+    case 'N': return visitor.visit_root_nan_atom(*this, value, TAPE_ERROR);
+    case 'i':
+    case 'I': return visitor.visit_root_inf_atom(*this, value);
+#else
     case 'n': return visitor.visit_root_null_atom(*this, value);
+#endif
     case '-':
     case '0': case '1': case '2': case '3': case '4':
     case '5': case '6': case '7': case '8': case '9':
@@ -21135,7 +28516,20 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::visit_primitive(V
   switch (*value) {
     case 't': return visitor.visit_true_atom(*this, value);
     case 'f': return visitor.visit_false_atom(*this, value);
+#if SIMDJSON_ENABLE_NAN_INF
+    case 'n': {
+      auto err = visitor.visit_null_atom(*this, value);
+      if (err == SUCCESS) { return err; }
+      // propagate the error value returned by a bad 'null' atom if parsing 'nan' fails
+      return visitor.visit_nan_atom(*this, value, err);
+    }
+    // 'N' isn't a canonically recognized atom, so we return a TAPE_ERROR if failure occurs
+    case 'N': return visitor.visit_nan_atom(*this, value, TAPE_ERROR);
+    case 'i':
+    case 'I': return visitor.visit_inf_atom(*this, value);
+#else
     case 'n': return visitor.visit_null_atom(*this, value);
+#endif
     default:
       log_error("Non-value found when value was expected!");
       return TAPE_ERROR;
@@ -21321,9 +28715,13 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_string(const uint8_t *src, u
            within the unicode codepoint handling code. */
         src += bs_dist;
         dst += bs_dist;
-        if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
-          return nullptr;
-        }
+        // Decode adjacent Unicode escapes without returning to the
+        // quote-and-backslash scanner between code points.
+        do {
+          if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
+            return nullptr;
+          }
+        } while (src[0] == '\\' && src[1] == 'u');
       } else {
         /* simple 1:1 conversion. Will eat bs_dist+2 characters in input and
          * write bs_dist+1 characters to output
@@ -21346,6 +28744,77 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_string(const uint8_t *src, u
   }
 }

+/**
+ * Bounds-safe variant of parse_string for input buffers that are NOT padded to
+ * len + SIMDJSON_PADDING bytes. `buf_end` is one past the last readable input
+ * byte (buf + len). It behaves exactly like parse_string while we are at least
+ * SIMDJSON_PADDING bytes away from buf_end (so every speculative SIMD read stays
+ * in bounds); once within the final SIMDJSON_PADDING bytes it copies the few
+ * remaining bytes into a space-padded scratch buffer and finishes with the
+ * regular parse_string. This keeps the delicate escape/Unicode handling in one
+ * place (the proven parse_string) rather than duplicating it.
+ *
+ * Correctness relies on stage 1 having validated the string, i.e. there is an
+ * unescaped closing quote within [src, buf_end); that quote is therefore inside
+ * the copied scratch, so parse_string finds it without running off the scratch.
+ */
+simdjson_warn_unused simdjson_inline uint8_t *parse_string_safe(const uint8_t *src, uint8_t *dst, bool allow_replacement, const uint8_t *buf_end) {
+  // Far from the end: identical to parse_string's loop. The guard uses
+  // SIMDJSON_PADDING (>= BYTES_PROCESSED) so copy_and_find never reads past
+  // buf_end; escape/Unicode look-aheads read within the string (before the
+  // closing quote, which is < buf_end), so they are in bounds here too.
+  // We add margin (+12) for handle_unicode_codepoint's worst-case lookahead:
+  // after seeing a high surrogate, it does hex_to_u32_nocheck on the immediate
+  // following bytes (+6 from the '\'), then (if it sees \u) another
+  // hex_to_u32_nocheck at +8..+11 relative to the backslash that started the
+  // escape. With bs_dist up to BYTES_PROCESSED-1 this reaches +11 from the
+  // chunk start. The +12 margin ensures that even on kernels where
+  // BYTES_PROCESSED == SIMDJSON_PADDING (e.g. icelake) the 4-byte read stays
+  // in-bounds. The scratch fallback (3*PAD) is already safe.
+  while (src + SIMDJSON_PADDING + 12 <= buf_end) {
+    auto b = backslash_and_quote{};
+    auto bs_quote = b.copy_and_find(src, dst);
+    if (bs_quote.has_quote_first()) {
+      return dst + bs_quote.quote_index();
+    }
+    if (bs_quote.has_backslash()) {
+      auto bs_dist = bs_quote.backslash_index();
+      uint8_t escape_char = src[bs_dist + 1];
+      if (escape_char == 'u') {
+        src += bs_dist;
+        dst += bs_dist;
+        if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
+          return nullptr;
+        }
+      } else {
+        uint8_t escape_result = escape_map[escape_char];
+        if (escape_result == 0u) {
+          return nullptr;
+        }
+        dst[bs_dist] = escape_result;
+        src += bs_dist + 2;
+        dst += bs_dist + 1;
+      }
+    } else {
+      src += backslash_and_quote::BYTES_PROCESSED;
+      dst += backslash_and_quote::BYTES_PROCESSED;
+    }
+  }
+  // Within the final SIMDJSON_PADDING bytes: copy what remains into a
+  // space-padded scratch (spaces are neither quote nor backslash, so they do not
+  // disturb matching) and let the regular parser finish from there. The closing
+  // quote is within `remaining` (< SIMDJSON_PADDING), so parse_string finds it in
+  // the chunk starting at some offset <= remaining and reads at most
+  // BYTES_PROCESSED (<= SIMDJSON_PADDING) further -- i.e. under 2*SIMDJSON_PADDING.
+  // We size at 3x for a comfortable margin (the unicode look-ahead reads a few
+  // extra bytes past an escape).
+  uint8_t scratch[SIMDJSON_PADDING * 3];
+  const size_t remaining = size_t(buf_end - src); // < SIMDJSON_PADDING
+  std::memset(scratch, ' ', sizeof(scratch));
+  std::memcpy(scratch, src, remaining);
+  return parse_string(scratch, dst, allow_replacement);
+}
+
 simdjson_warn_unused simdjson_inline uint8_t *parse_wobbly_string(const uint8_t *src, uint8_t *dst) {
   // It is not ideal that this function is nearly identical to parse_string.
   while (1) {
@@ -21400,73 +28869,6 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_wobbly_string(const uint8_t

 #endif // SIMDJSON_SRC_GENERIC_STAGE2_STRINGPARSING_H
 /* end file generic/stage2/stringparsing.h for haswell */
-/* including generic/stage2/structural_iterator.h for haswell: #include <generic/stage2/structural_iterator.h> */
-/* begin file generic/stage2/structural_iterator.h for haswell */
-#ifndef SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H
-
-/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
-/* amalgamation skipped (editor-only): #define SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H */
-/* amalgamation skipped (editor-only): #include <generic/stage2/base.h> */
-/* amalgamation skipped (editor-only): #include <simdjson/generic/dom_parser_implementation.h> */
-/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
-
-namespace simdjson {
-namespace haswell {
-namespace {
-namespace stage2 {
-
-class structural_iterator {
-public:
-  const uint8_t* const buf;
-  uint32_t *next_structural;
-  dom_parser_implementation &dom_parser;
-
-  // Start a structural
-  simdjson_inline structural_iterator(dom_parser_implementation &_dom_parser, size_t start_structural_index)
-    : buf{_dom_parser.buf},
-      next_structural{&_dom_parser.structural_indexes[start_structural_index]},
-      dom_parser{_dom_parser} {
-  }
-  // Get the buffer position of the current structural character
-  simdjson_inline const uint8_t* current() {
-    return &buf[*(next_structural-1)];
-  }
-  // Get the current structural character
-  simdjson_inline char current_char() {
-    return buf[*(next_structural-1)];
-  }
-  // Get the next structural character without advancing
-  simdjson_inline char peek_next_char() {
-    return buf[*next_structural];
-  }
-  simdjson_inline const uint8_t* peek() {
-    return &buf[*next_structural];
-  }
-  simdjson_inline const uint8_t* advance() {
-    return &buf[*(next_structural++)];
-  }
-  simdjson_inline char advance_char() {
-    return buf[*(next_structural++)];
-  }
-  simdjson_inline size_t remaining_len() {
-    return dom_parser.len - *(next_structural-1);
-  }
-
-  simdjson_inline bool at_end() {
-    return next_structural == &dom_parser.structural_indexes[dom_parser.n_structural_indexes];
-  }
-  simdjson_inline bool at_beginning() {
-    return next_structural == dom_parser.structural_indexes.get();
-  }
-};
-
-} // namespace stage2
-} // unnamed namespace
-} // namespace haswell
-} // namespace simdjson
-
-#endif // SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H
-/* end file generic/stage2/structural_iterator.h for haswell */
 /* including generic/stage2/tape_builder.h for haswell: #include <generic/stage2/tape_builder.h> */
 /* begin file generic/stage2/tape_builder.h for haswell */
 #ifndef SIMDJSON_SRC_GENERIC_STAGE2_TAPE_BUILDER_H
@@ -21489,12 +28891,8 @@ namespace haswell {
 namespace {
 namespace stage2 {

-struct tape_builder {
-  template<bool STREAMING>
-  simdjson_warn_unused static simdjson_inline error_code parse_document(
-    dom_parser_implementation &dom_parser,
-    dom::document &doc) noexcept;
-
+template <bool UNPADDED>
+struct tape_builder_impl {
   /** Called when a non-empty document starts. */
   simdjson_warn_unused simdjson_inline error_code visit_document_start(json_iterator &iter) noexcept;
   /** Called when a non-empty document ends without error. */
@@ -21547,88 +28945,130 @@ struct tape_builder {
   simdjson_warn_unused simdjson_inline error_code visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept;
   simdjson_warn_unused simdjson_inline error_code visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept;

+#if SIMDJSON_ENABLE_NAN_INF
+  simdjson_warn_unused simdjson_inline error_code visit_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept;
+  simdjson_warn_unused simdjson_inline error_code visit_root_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept;
+  // Attempts to parse 'inf' or 'infinity' (case insensitive). Because neither are canonical atoms,
+  // this returns a tape error on failure.
+  simdjson_warn_unused simdjson_inline error_code visit_inf_atom(json_iterator &iter, const uint8_t *value) noexcept;
+  simdjson_warn_unused simdjson_inline error_code visit_root_inf_atom(json_iterator &iter, const uint8_t *value) noexcept;
+#endif
+
   /** Called each time a new field or element in an array or object is found. */
   simdjson_warn_unused simdjson_inline error_code increment_count(json_iterator &iter) noexcept;

   /** Next location to write to tape */
   tape_writer tape;
+public:
+  simdjson_inline tape_builder_impl(dom::document &doc) noexcept;
 private:
   /** Next write location in the string buf for stage 2 parsing */
   uint8_t *current_string_buf_loc;

-  simdjson_inline tape_builder(dom::document &doc) noexcept;
-
   simdjson_inline uint32_t next_tape_index(json_iterator &iter) const noexcept;
   simdjson_inline void start_container(json_iterator &iter) noexcept;
   simdjson_warn_unused simdjson_inline error_code end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept;
   simdjson_warn_unused simdjson_inline error_code empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept;
   simdjson_inline uint8_t *on_start_string(json_iterator &iter) noexcept;
   simdjson_inline void on_end_string(uint8_t *dst) noexcept;
-}; // struct tape_builder
+}; // struct tape_builder_impl

-template<bool STREAMING>
-simdjson_warn_unused simdjson_inline error_code tape_builder::parse_document(
-    dom_parser_implementation &dom_parser,
-    dom::document &doc) noexcept {
-  dom_parser.doc = &doc;
-  json_iterator iter(dom_parser, STREAMING ? dom_parser.next_structural_index : 0);
-  tape_builder builder(doc);
-  return iter.walk_document<STREAMING>(builder);
-}
+// Thin, non-templated entry so each architecture's stage2() keeps calling
+// tape_builder::parse_document<STREAMING> unchanged. It chooses the bounds-safe
+// (unpadded) or the regular (padded) tape_builder_impl ONCE per document, so the
+// choice is a compile-time constant inside the walk: the padded path carries no
+// extra branch or load (see tape_builder_impl::visit_string).
+struct tape_builder {
+  template<bool STREAMING>
+  simdjson_warn_unused static simdjson_inline error_code parse_document(
+      dom_parser_implementation &dom_parser, dom::document &doc) noexcept {
+    dom_parser.doc = &doc;
+    json_iterator iter(dom_parser, STREAMING ? dom_parser.next_structural_index : 0);
+    if (dom_parser._unpadded) {
+      tape_builder_impl<true> builder(doc);
+      return iter.walk_document<STREAMING, true>(builder);
+    } else {
+      tape_builder_impl<false> builder(doc);
+      return iter.walk_document<STREAMING, false>(builder);
+    }
+  }
+};

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_primitive(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_primitive(json_iterator &iter, const uint8_t *value) noexcept {
   return iter.visit_root_primitive(*this, value);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_primitive(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_primitive(json_iterator &iter, const uint8_t *value) noexcept {
   return iter.visit_primitive(*this, value);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_empty_object(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_empty_object(json_iterator &iter) noexcept {
   return empty_container(iter, internal::tape_type::START_OBJECT, internal::tape_type::END_OBJECT);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_empty_array(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_empty_array(json_iterator &iter) noexcept {
   return empty_container(iter, internal::tape_type::START_ARRAY, internal::tape_type::END_ARRAY);
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_document_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_document_start(json_iterator &iter) noexcept {
   start_container(iter);
   return SUCCESS;
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_object_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_object_start(json_iterator &iter) noexcept {
   start_container(iter);
   return SUCCESS;
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_array_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_array_start(json_iterator &iter) noexcept {
   start_container(iter);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_object_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_object_end(json_iterator &iter) noexcept {
   return end_container(iter, internal::tape_type::START_OBJECT, internal::tape_type::END_OBJECT);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_array_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_array_end(json_iterator &iter) noexcept {
   return end_container(iter, internal::tape_type::START_ARRAY, internal::tape_type::END_ARRAY);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_document_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_document_end(json_iterator &iter) noexcept {
   constexpr uint32_t start_tape_index = 0;
   tape.append(start_tape_index, internal::tape_type::ROOT);
   tape_writer::write(iter.dom_parser.doc->tape[start_tape_index], next_tape_index(iter), internal::tape_type::ROOT);
   return SUCCESS;
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_key(json_iterator &iter, const uint8_t *key) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_key(json_iterator &iter, const uint8_t *key) noexcept {
   return visit_string(iter, key, true);
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::increment_count(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::increment_count(json_iterator &iter) noexcept {
   iter.dom_parser.open_containers[iter.depth].count++; // we have a key value pair in the object at parser.dom_parser.depth - 1
   return SUCCESS;
 }

-simdjson_inline tape_builder::tape_builder(dom::document &doc) noexcept : tape{doc.tape.get()}, current_string_buf_loc{doc.string_buf.get()} {}
+template <bool UNPADDED>
+simdjson_inline tape_builder_impl<UNPADDED>::tape_builder_impl(dom::document &doc) noexcept : tape{doc.tape.get()}, current_string_buf_loc{doc.string_buf.get()} {}

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_string(json_iterator &iter, const uint8_t *value, bool key) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_string(json_iterator &iter, const uint8_t *value, bool key) noexcept {
   iter.log_value(key ? "key" : "string");
   uint8_t *dst = on_start_string(iter);
-  dst = stringparsing::parse_string(value+1, dst, false); // We do not allow replacement when the escape characters are invalid.
+  // We do not allow replacement when the escape characters are invalid.
+  // UNPADDED is a compile-time constant chosen once per document by
+  // tape_builder::parse_document, so the padded build instantiates only the
+  // plain parse_string call below -- no runtime branch and no flag load.
+  SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+    dst = stringparsing::parse_string_safe(value+1, dst, false, iter.buf + iter.dom_parser.len);
+  } else {
+    dst = stringparsing::parse_string(value+1, dst, false);
+  }
   if (dst == nullptr) {
     iter.log_error("Invalid escape in string");
     return STRING_ERROR;
@@ -21637,27 +29077,48 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_string(json_
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_string(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_string(json_iterator &iter, const uint8_t *value) noexcept {
   return visit_string(iter, value);
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_number(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_number(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("number");
-  error_code err = numberparsing::parse_number(value, tape);
+  const uint8_t *num = value;
+  std::unique_ptr<uint8_t[]> copy{}; // keeps a padded copy of the tail alive when used
+  SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+    // numberparsing reads ahead in 8-byte blocks for floats
+    // (is_made_of_eight_digits_fast reads up to 7 bytes past the digits), so a
+    // number whose digits reach the final bytes of an unpadded buffer would read
+    // past it. *(next_structural) is the offset of the token following this
+    // number, hence an upper bound on where the digits end; when that is within
+    // SIMDJSON_PADDING of the end we parse from a space-padded copy of the tail
+    // (mirroring visit_root_number). This fires only for numbers near the end.
+    if (simdjson_unlikely(*(iter.next_structural) + SIMDJSON_PADDING > iter.dom_parser.len)) {
+      const size_t rl = iter.remaining_len(); // bytes from `value` to the end of the document
+      copy.reset(new (std::nothrow) uint8_t[rl + SIMDJSON_PADDING]);
+      if (copy.get() == nullptr) { return MEMALLOC; }
+      std::memcpy(copy.get(), value, rl);
+      std::memset(copy.get() + rl, ' ', SIMDJSON_PADDING);
+      num = copy.get();
+    }
+  }
+  error_code err = numberparsing::parse_number(num, tape);
   if (simdjson_unlikely(err == BIGINT_ERROR &&
       iter.dom_parser._number_as_string)) {
     // Write big integer to string buffer using the same format as strings.
     // Scan digits the same way parse_number does (skip optional '-', then digits).
-    const uint8_t *p = value;
+    const uint8_t *p = num;
     if (*p == '-') p++;
     while (numberparsing::is_digit(*p)) p++;
     // The digit run must be terminated by a structural or whitespace character; otherwise the
     // token is malformed (e.g. "123456789123456789123x").
     if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
-    size_t len = size_t(p - value);
+    size_t len = size_t(p - num);
     tape.append(current_string_buf_loc - iter.dom_parser.doc->string_buf.get(), internal::tape_type::BIGINT);
     uint8_t *dst = current_string_buf_loc + sizeof(uint32_t);
-    memcpy(dst, value, len);
+    memcpy(dst, num, len);
     dst += len;
     on_end_string(dst);
     return SUCCESS;
@@ -21665,7 +29126,8 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_number(json_
   return err;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_number(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_number(json_iterator &iter, const uint8_t *value) noexcept {
   //
   // We need to make a copy to make sure that the string is space terminated.
   // This is not about padding the input, which should already padded up
@@ -21679,76 +29141,142 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_number(
   // practice unless you are in the strange scenario where you have many JSON
   // documents made of single atoms.
   //
-  std::unique_ptr<uint8_t[]>copy(new (std::nothrow) uint8_t[iter.remaining_len() + SIMDJSON_PADDING]);
+  // In a stream, the input goes on with other documents: copy up to the next
+  // structural only, not to the end of the batch.
+  const size_t len = (std::min)(iter.remaining_len(), size_t(*iter.next_structural) - size_t(*(iter.next_structural - 1)));
+  std::unique_ptr<uint8_t[]>copy(new (std::nothrow) uint8_t[len + SIMDJSON_PADDING]);
   if (copy.get() == nullptr) { return MEMALLOC; }
-  std::memcpy(copy.get(), value, iter.remaining_len());
-  std::memset(copy.get() + iter.remaining_len(), ' ', SIMDJSON_PADDING);
+  std::memcpy(copy.get(), value, len);
+  std::memset(copy.get() + len, ' ', SIMDJSON_PADDING);
   error_code error = visit_number(iter, copy.get());
   return error;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("true");
-  if (!atomparsing::is_valid_true_atom(value)) { return T_ATOM_ERROR; }
+  // The non-length-aware validator reads a fixed 5 bytes; a malformed/truncated
+  // token at the very end of an unpadded buffer would over-read. Use the
+  // length-aware form there (the root variant already does this).
+  const bool ok = UNPADDED ? atomparsing::is_valid_true_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_true_atom(value);
+  if (!ok) { return T_ATOM_ERROR; }
   tape.append(0, internal::tape_type::TRUE_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("true");
   if (!atomparsing::is_valid_true_atom(value, iter.remaining_len())) { return T_ATOM_ERROR; }
   tape.append(0, internal::tape_type::TRUE_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("false");
-  if (!atomparsing::is_valid_false_atom(value)) { return F_ATOM_ERROR; }
+  const bool ok = UNPADDED ? atomparsing::is_valid_false_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_false_atom(value);
+  if (!ok) { return F_ATOM_ERROR; }
   tape.append(0, internal::tape_type::FALSE_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("false");
   if (!atomparsing::is_valid_false_atom(value, iter.remaining_len())) { return F_ATOM_ERROR; }
   tape.append(0, internal::tape_type::FALSE_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("null");
-  if (!atomparsing::is_valid_null_atom(value)) { return N_ATOM_ERROR; }
+  const bool ok = UNPADDED ? atomparsing::is_valid_null_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_null_atom(value);
+  if (!ok) { return N_ATOM_ERROR; }
   tape.append(0, internal::tape_type::NULL_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("null");
   if (!atomparsing::is_valid_null_atom(value, iter.remaining_len())) { return N_ATOM_ERROR; }
   tape.append(0, internal::tape_type::NULL_VALUE);
   return SUCCESS;
 }

+#if SIMDJSON_ENABLE_NAN_INF
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept {
+  iter.log_value("nan");
+  // For unpadded input use the length-aware validator so the 'infinity'-style
+  // 8-byte compare cannot read past the buffer on a malformed token at the end.
+  const bool ok = UNPADDED ? atomparsing::is_valid_nan_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_nan_atom(value);
+  if (!ok) { return errc; }
+  tape.append_double(std::numeric_limits<double>::quiet_NaN());
+  return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept {
+  iter.log_value("nan");
+  if (!atomparsing::is_valid_nan_atom(value, iter.remaining_len())) { return errc; }
+  tape.append_double(std::numeric_limits<double>::quiet_NaN());
+  return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_inf_atom(json_iterator &iter, const uint8_t *value) noexcept {
+  iter.log_value("inf");
+  // Because 'inf' is an extension, non a canonical atom, a tape error should be returned on failure.
+  // For unpadded input use the length-aware validator so the 'infinity' 8-byte
+  // compare cannot read past the buffer on a malformed token at the end.
+  const bool ok = UNPADDED ? atomparsing::is_valid_inf_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_inf_atom(value);
+  if (!ok) { return TAPE_ERROR; }
+  tape.append_double(std::numeric_limits<double>::infinity());
+  return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_inf_atom(json_iterator &iter, const uint8_t *value) noexcept {
+  iter.log_value("inf");
+  // Because 'inf' is an extension, non a canonical atom, a tape error should be returned on failure
+  if (!atomparsing::is_valid_inf_atom(value, iter.remaining_len())) { return TAPE_ERROR; }
+  tape.append_double(std::numeric_limits<double>::infinity());
+  return SUCCESS;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
 // private:

-simdjson_inline uint32_t tape_builder::next_tape_index(json_iterator &iter) const noexcept {
+template <bool UNPADDED>
+simdjson_inline uint32_t tape_builder_impl<UNPADDED>::next_tape_index(json_iterator &iter) const noexcept {
   return uint32_t(tape.next_tape_loc - iter.dom_parser.doc->tape.get());
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
   auto start_index = next_tape_index(iter);
   tape.append(start_index+2, start);
   tape.append(start_index, end);
   return SUCCESS;
 }

-simdjson_inline void tape_builder::start_container(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_inline void tape_builder_impl<UNPADDED>::start_container(json_iterator &iter) noexcept {
   iter.dom_parser.open_containers[iter.depth].tape_index = next_tape_index(iter);
   iter.dom_parser.open_containers[iter.depth].count = 0;
   tape.skip(); // We don't actually *write* the start element until the end.
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
   // Write the ending tape element, pointing at the start location
   const uint32_t start_tape_index = iter.dom_parser.open_containers[iter.depth].tape_index;
   tape.append(start_tape_index, end);
@@ -21761,13 +29289,15 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::end_container(json
   return SUCCESS;
 }

-simdjson_inline uint8_t *tape_builder::on_start_string(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_inline uint8_t *tape_builder_impl<UNPADDED>::on_start_string(json_iterator &iter) noexcept {
   // we advance the point, accounting for the fact that we have a NULL termination
   tape.append(current_string_buf_loc - iter.dom_parser.doc->string_buf.get(), internal::tape_type::STRING);
   return current_string_buf_loc + sizeof(uint32_t);
 }

-simdjson_inline void tape_builder::on_end_string(uint8_t *dst) noexcept {
+template <bool UNPADDED>
+simdjson_inline void tape_builder_impl<UNPADDED>::on_end_string(uint8_t *dst) noexcept {
   uint32_t str_length = uint32_t(dst - (current_string_buf_loc + sizeof(uint32_t)));
   // TODO check for overflow in case someone has a crazy string (>=4GB?)
   // But only add the overflow check when the document itself exceeds 4GB
@@ -22123,16 +29653,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
 }
 #endif

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
-                                uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
-  return _addcarry_u64(0, value1, value2,
-                       reinterpret_cast<unsigned __int64 *>(result));
-#else
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-#endif
-}

 } // unnamed namespace
 } // namespace icelake
@@ -22888,6 +30408,9 @@ namespace atomparsing {
 // to the compile-time constant 1936482662.
 simdjson_inline uint32_t string_to_uint32(const char* str) { uint32_t val; std::memcpy(&val, str, sizeof(uint32_t)); return val; }

+// Acts on the same principle as string_to_uint32, but on an 8-byte block of memory
+simdjson_inline uint64_t string_to_uint64(const char* str) { uint64_t val; std::memcpy(&val, str, sizeof(uint64_t)); return val; }
+

 // Again in str4ncmp we use a memcpy to avoid undefined behavior. The memcpy may appear expensive.
 // Yet all decent optimizing compilers will compile memcpy to a single instruction, just about.
@@ -22899,6 +30422,28 @@ simdjson_inline uint32_t str4ncmp(const uint8_t *src, const char* atom) {
   return srcval ^ string_to_uint32(atom);
 }

+// Checks that the first 8 characters of the input string match the given atom in a case-insensitive manner.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint64_t str8ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+  uint64_t srcval; // we want to avoid unaligned 32-bit loads (undefined in C/C++)
+  static_assert(sizeof(uint64_t) <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be larger than 8 bytes");
+  std::memcpy(&srcval, src, sizeof(uint64_t));
+
+  return (srcval | 0x2020202020202020ull) ^ string_to_uint64(atom);
+}
+
+// Checks that the first 3 characters of 'src' match 'atom' in a case-insensitive way.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint32_t str3ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+  return ((src[0] | 0x20) ^ atom[0]) //
+       | ((src[1] | 0x20) ^ atom[1]) //
+       | ((src[2] | 0x20) ^ atom[2]);
+}
+
 simdjson_warn_unused
 simdjson_inline bool is_valid_true_atom(const uint8_t *src) {
   return (str4ncmp(src, "true") | jsoncharutils::is_not_structural_or_whitespace(src[4])) == 0;
@@ -22935,6 +30480,71 @@ simdjson_inline bool is_valid_null_atom(const uint8_t *src, size_t len) {
   else { return false; }
 }

+#if SIMDJSON_ENABLE_NAN_INF
+// "nan" is 3 bytes; we check characters and then verify the next
+// character is structural or whitespace. We accept both "nan" and "NaN".
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src) {
+  return (str3ncmp_case_insensitive(src, "nan")
+        | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+}
+
+// checks that the next four characters of a string are 'nan"', where the 'nan'
+// is checked in a case-insensitive way.
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_in_string(const uint8_t *src) {
+  return (str3ncmp_case_insensitive(src, "nan") | (src[3] ^ '"')) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src, size_t len) {
+  if (len > 3) { return is_valid_nan_atom(src); }
+  if (len == 3) { return str3ncmp_case_insensitive(src, "nan") == 0; }
+  return false;
+}
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src) {
+  bool is_short_inf = (str3ncmp_case_insensitive(src, "inf")
+                     | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+  if(is_short_inf) return true;
+
+  // Check for 'infinity' (any capitalization)
+  return (str8ncmp_case_insensitive(src, "infinity") | jsoncharutils::is_not_structural_or_whitespace(src[8])) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_in_string(const uint8_t *src) {
+  bool is_short_inf = (str3ncmp_case_insensitive(src, "inf") | (src[3] ^ '"')) == 0;
+  if(is_short_inf) return true;
+
+  return (str8ncmp_case_insensitive(src, "infinity") | (src[8] ^ '"')) == 0;
+}
+
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src, size_t len) {
+  if (len > 8) { return is_valid_inf_atom(src); }
+  if (len == 8 && str8ncmp_case_insensitive(src, "infinity") == 0) {
+    return true;
+  }
+  if (len > 3) {
+    return (str3ncmp_case_insensitive(src, "inf")
+          | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+  }
+  if (len == 3) { return str3ncmp_case_insensitive(src, "inf") == 0; }
+  return false;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
 } // namespace atomparsing
 } // unnamed namespace
 } // namespace icelake
@@ -23025,7 +30635,7 @@ inline simdjson_warn_unused error_code dom_parser_implementation::set_capacity(s
 }

 inline simdjson_warn_unused error_code dom_parser_implementation::set_max_depth(size_t max_depth) noexcept {
-  if(max_depth == 0 || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
+  if(max_depth == 0 || max_depth > SIMDJSON_MAX_DEPTH || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
   // Stage 2 stacks
   open_containers.reset(new (std::nothrow) open_container[max_depth]);
   is_array.reset(new (std::nothrow) bool[max_depth]);
@@ -23204,6 +30814,7 @@ protected:
 /* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_NUMBERPARSING_H */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/base.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/jsoncharutils.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/atomparsing.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/internal/numberparsing_tables.h" */
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */

@@ -23243,6 +30854,18 @@ simdjson_inline double to_double(uint64_t mantissa, uint64_t real_exponent, bool
     return d;
 }

+// Convert a mantissa, an exponent and a sign bit into an ieee32 float (binary32).
+// The real_exponent needs to be in [0, 254] (technically real_exponent = 255 would be acceptable).
+// The mantissa should be in [0,1<<24). The bit at index (1U << 23) will be zeroed.
+simdjson_inline float to_float(uint32_t mantissa, uint32_t real_exponent, bool negative) {
+    float f;
+    mantissa &= ~(uint32_t(1) << 23);
+    mantissa |= real_exponent << 23;
+    mantissa |= ((static_cast<uint32_t>(negative)) << 31);
+    std::memcpy(&f, &mantissa, sizeof(f));
+    return f;
+}
+
 // Attempts to compute i * 10^(power) exactly; and if "negative" is
 // true, negate the result.
 // This function will only work in some cases, when it does not work, success is
@@ -23499,6 +31122,219 @@ simdjson_inline bool compute_float_64(int64_t power, uint64_t i, bool negative,
   return true;
 }

+// Attempts to compute i * 10^(power) as a binary32 (float) value; and if
+// "negative" is true, negate the result.
+//
+// This is the same Eisel-Lemire algorithm as compute_float_64, retargeted at
+// binary32. It is adapted from fast_float
+// (https://github.com/fastfloat/fast_float), where the routine is written once
+// and instantiated for each binary format. We only need 24 bits of mantissa
+// instead of 53, so we ask the truncated product for 24 + 2 = 26 accurate bits
+// (plus one bit that may be lost to the "upperbit" shift), and we shift the
+// product right by (upperbit + 64 - 23 - 3) instead of (upperbit + 64 - 52 - 3).
+//
+// The power-of-five table is shared with the binary64 code: the useful range
+// for binary32, [smallest_power_binary32, largest_power_binary32], is a strict
+// subset of [smallest_power, largest_power].
+//
+// The function returns false when the result would be infinite: simdjson
+// refuses to parse infinite values, so the caller reports an error. Unlike
+// compute_float_64, a false return never means "try harder": the accuracy of
+// the two-limb product is guaranteed by Noble Mushtak and Daniel Lemire,
+// "Fast Number Parsing Without Fallback" (https://arxiv.org/abs/2212.06644).
+// The caller must still fall back to a slow path when the 64-bit mantissa i
+// was truncated (more than 19 significant digits).
+//
+// We assume that power is in the [smallest_power, largest_power] interval: the
+// caller is responsible for this check.
+simdjson_inline bool compute_float_32(int64_t power, uint64_t i, bool negative, float &d) {
+  // Powers of ten that are exactly representable as binary32 values: 10^k is
+  // exact as long as k <= 10 (5^10 = 9765625 < 2**24 while 5^11 > 2**24).
+  static constexpr float power_of_ten_float[] = {
+      1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+      1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+  // The range of powers of ten that a non-zero, finite binary32 value can be
+  // built from. Anything smaller rounds to zero, anything larger is infinite.
+  // These are the smallest_power_of_ten()/largest_power_of_ten() constants that
+  // fast_float uses for binary32.
+  constexpr int smallest_power_binary32 = -65;
+  constexpr int largest_power_binary32 = 38;
+
+  // We start with the fast path described in
+  // Clinger WD. How to read floating point numbers accurately.
+  // ACM SIGPLAN Notices. 1990
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+  // We cannot be certain that x/y is rounded to nearest.
+  if (0 <= power && power <= 10 && i <= 16777215)
+#else
+  if (-10 <= power && power <= 10 && i <= 16777215)
+#endif
+  {
+    // Convert the integer into a float. This is lossless since
+    // 0 <= i <= 2^24 - 1.
+    d = float(i);
+    // Both d and the power of ten are exactly representable as binary32
+    // values, so the product (or quotient) is correctly rounded.
+    if (power < 0) {
+      d = d / power_of_ten_float[-power];
+    } else {
+      d = d * power_of_ten_float[power];
+    }
+    if (negative) {
+      d = -d;
+    }
+    return true;
+  }
+
+  // The fast path has failed, so we fall back on the Eisel-Lemire algorithm.
+  // It needs i > 0 (so that the leading bit of i can be normalized), so we
+  // handle i == 0 separately. We also handle the powers of ten that are so
+  // small (or so large) that the answer is zero (or infinite) whatever the
+  // mantissa is.
+  if (i == 0 || power < smallest_power_binary32) {
+    d = negative ? -0.0f : 0.0f;
+    return true;
+  }
+  if (power > largest_power_binary32) {
+    // We have, for sure, an infinite value.
+    return false;
+  }
+
+  // The exponent is 128 + 63 + power + floor(log(5**power)/log(2)).
+  // The 128 comes from the ieee32 standard (the minimal exponent is -127).
+  // The 63 comes from the fact that we use a 64-bit word.
+  // See compute_float_64 for a discussion of the magical 152170 + 65536.
+  int64_t exponent = (((152170 + 65536) * power) >> 16) + 128 + 63;
+
+  // We want the most significant bit of i to be 1. Shift if needed.
+  int lz = leading_zeroes(i);
+  i <<= lz;
+
+  // We want the most significant 64 bits of the product i * 5**power. It is
+  // safe to index the table because
+  // smallest_power <= smallest_power_binary32 <= power
+  //               <= largest_power_binary32 <= largest_power.
+  const uint32_t index = 2 * uint32_t(power - simdjson::internal::smallest_power);
+#if SIMDJSON_STATIC_REFLECTION
+  simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index]);
+#else
+  simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index]);
+#endif
+
+  // Unless the least significant 38 bits of the high (64-bit) part of the full
+  // product are all 1s, then we know that the most significant 26 bits are
+  // exact and no further work is needed. Having 26 bits is necessary because
+  // we need 24 bits for the mantissa but we have to have one rounding bit and
+  // we can waste a bit if the most significant bit of the product is zero.
+  // (For binary64, the same reasoning gives 55 bits and a 9-bit mask.)
+  if((firstproduct.high & 0x3FFFFFFFFF) == 0x3FFFFFFFFF) {
+    // The truncated multiplication was not accurate enough; use the next 64
+    // bits of the power of five to refine it. See compute_float_64 for a
+    // detailed discussion.
+#if SIMDJSON_STATIC_REFLECTION
+    simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index + 1]);
+#else
+    simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index + 1]);
+#endif
+    firstproduct.low += secondproduct.high;
+    if(secondproduct.high > firstproduct.low) { firstproduct.high++; }
+  }
+  uint64_t lower = firstproduct.low;
+  uint64_t upper = firstproduct.high;
+  // The final mantissa should be 24 bits with a leading 1.
+  // We shift it so that it occupies 25 bits with a leading 1.
+  ///////
+  uint64_t upperbit = upper >> 63;
+  uint64_t mantissa = upper >> (upperbit + 38); // 38 == 64 - 23 - 3
+  lz += int(1 ^ upperbit);
+
+  // Here we have mantissa < (1<<25).
+  int64_t real_exponent = exponent - lz;
+  if (simdjson_unlikely(real_exponent <= 0)) { // we have a subnormal?
+    // Here we have that real_exponent <= 0 so -real_exponent >= 0
+    if(-real_exponent + 1 >= 64) { // if we have more than 64 bits below the minimum exponent, you have a zero for sure.
+      d = negative ? -0.0f : 0.0f;
+      return true;
+    }
+    // next line is safe because -real_exponent + 1 < 64
+    mantissa >>= -real_exponent + 1;
+    // Thankfully, we can't have both "round-to-even" and subnormals because
+    // "round-to-even" only occurs for powers close to 0.
+    mantissa += (mantissa & 1); // round up
+    mantissa >>= 1;
+    // As in compute_float_64, rounding up may take us out of the subnormal
+    // range, so we can only decide after rounding.
+    real_exponent = (mantissa < (uint64_t(1) << 23)) ? 0 : 1;
+    d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+    return true;
+  }
+  // We have to round to even. The "to even" part is only a problem when we are
+  // right in between two floats, which we guard against. The bounds on the
+  // power of ten are those of fast_float's
+  // min_exponent_round_to_even()/max_exponent_round_to_even() for binary32:
+  // when q >= 0, (2m+1) must be divisible by 5^q with 5^q <= 2^25, so q <= 10;
+  // when q < 0, we need 2^24 x 5^{-q} < 2^{64}, so q >= -17.
+  if (simdjson_unlikely((lower <= 1) && (power >= -17) && (power <= 10) && ((mantissa & 3) == 1))) {
+    if((mantissa << (upperbit + 38)) == upper) {
+      mantissa &= ~uint64_t(1);   // flip it so that we do not round up
+    }
+  }
+
+  mantissa += mantissa & 1;
+  mantissa >>= 1;
+
+  // Here we have mantissa < (1<<24), unless there was an overflow
+  if (mantissa >= (uint64_t(1) << 24)) {
+    mantissa = (uint64_t(1) << 23);
+    real_exponent++;
+  }
+  mantissa &= ~(uint64_t(1) << 23);
+  // we have to check that real_exponent is in range, otherwise we bail out
+  if (simdjson_unlikely(real_exponent > 254)) {
+    // We have an infinite value!!! We could actually throw an error here if we could.
+    return false;
+  }
+  d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+  return true;
+}
+
+#if SIMDJSON_ENABLE_NAN_INF
+// Parses a nan or infinity. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, double& d) noexcept {
+  if (atomparsing::is_valid_inf_atom(src)) {
+    double inf = std::numeric_limits<double>::infinity();
+    d = negative ? -inf : inf;
+    return true;
+  }
+
+  if (atomparsing::is_valid_nan_atom(src)) {
+    d = std::numeric_limits<double>::quiet_NaN();
+    return true;
+  }
+
+  return false;
+}
+
+// Parses a nan or infinity as a binary32 value. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, float& d) noexcept {
+  if (atomparsing::is_valid_inf_atom(src)) {
+    float inf = std::numeric_limits<float>::infinity();
+    d = negative ? -inf : inf;
+    return true;
+  }
+
+  if (atomparsing::is_valid_nan_atom(src)) {
+    d = std::numeric_limits<float>::quiet_NaN();
+    return true;
+  }
+
+  return false;
+}
+#endif
+
 // We call a fallback floating-point parser that might be slow. Note
 // it will accept JSON numbers, but the JSON spec. is more restrictive so
 // before you call parse_float_fallback, you need to have validated the input
@@ -23536,6 +31372,16 @@ static bool parse_float_fallback(const uint8_t *ptr, const uint8_t *end_ptr, dou
   return !(*outDouble > (std::numeric_limits<double>::max)() || *outDouble < std::numeric_limits<double>::lowest());
 }

+// Same as parse_float_fallback, but for binary32 (float) values. Going through
+// the binary64 fallback and then rounding to binary32 would be subject to
+// double rounding, so we run the fallback algorithm directly on binary32.
+static bool parse_float_fallback(const uint8_t *ptr, float *outFloat) {
+  *outFloat = simdjson::internal::from_chars_float(reinterpret_cast<const char *>(ptr));
+  // We do not accept infinite values. See the binary64 version above for why we
+  // do not use std::isfinite.
+  return !(*outFloat > (std::numeric_limits<float>::max)() || *outFloat < std::numeric_limits<float>::lowest());
+}
+
 // check quickly whether the next 8 chars are made of digits
 // at a glance, it looks better than Mula's
 // http://0x80.pl/articles/swar-digits-validate.html
@@ -23554,6 +31400,28 @@ simdjson_inline bool is_made_of_eight_digits_fast(const uint8_t *chars) {
           0x3333333333333333);
 }

+// The same idea for four characters. A block of eight only bites when eight
+// digits are left, so a run of 4 to 7 digits used to fall back to one digit at a
+// time; taking four of them at once is what makes the tail of a long fraction
+// cheap. Adapted from fast_float, credit @aqrit.
+simdjson_inline bool is_made_of_four_digits_fast(const uint8_t *chars) {
+  uint32_t val;
+  // Reads up to 3 bytes beyond the digits, which SIMDJSON_PADDING covers.
+  static_assert(3 <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be bigger than 3");
+  std::memcpy(&val, chars, 4);
+  return !(((val + 0x46464646) | (val - 0x30303030)) & 0x80808080);
+}
+
+// Only call this when is_made_of_four_digits_fast() says the four characters are
+// digits. Little-endian: guarded by SIMDJSON_SWAR_NUMBER_PARSING.
+simdjson_inline uint32_t parse_four_digits_unrolled(const uint8_t *chars) {
+  uint32_t val;
+  std::memcpy(&val, chars, 4);
+  val -= 0x30303030;
+  val = (val * 10) + (val >> 8);
+  return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
 template<typename I>
 SIMDJSON_NO_SANITIZE_UNDEFINED // We deliberately allow overflow here and check later
 simdjson_inline bool parse_digit(const uint8_t c, I &i) {
@@ -23570,26 +31438,73 @@ simdjson_inline bool is_digit(const uint8_t c) {
   return static_cast<uint8_t>(c - '0') <= 9;
 }

-simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
-  // we continue with the fiction that we have an integer. If the
-  // floating point number is representable as x * 10^z for some integer
-  // z that fits in 53 bits, then we will be able to convert back the
-  // the integer into a float in a lossless manner.
-  const uint8_t *const first_after_period = p;
+// Consumes a run of digits into i, eight at a time while eight are available,
+// then four, then one at a time. Overflow is deliberate: the caller counts the
+// digits and falls back when there are too many for a 64-bit mantissa.
+//
+// Long runs of digits are what a float-heavy document is made of: the fraction
+// of a binary64 printed to full precision is 15 to 17 digits, and reading those
+// one at a time is the single largest cost in parse_double(). Integer parts are
+// left alone: they are usually a handful of digits, and there the check for a
+// block of eight is wasted work.
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_fraction_digits(const uint8_t *&p, uint64_t &i) {
+#ifdef SIMDJSON_SWAR_NUMBER_PARSING
+#if SIMDJSON_SWAR_NUMBER_PARSING
+  while (is_made_of_eight_digits_fast(p)) {
+    i = i * 100000000 + parse_eight_digits_unrolled(p);
+    p += 8;
+  }
+  // A 4 to 7 digit remainder is the common case once the blocks of eight are
+  // gone: 15 fraction digits leave 7, and 17 leave 1 after two blocks.
+  if (is_made_of_four_digits_fast(p)) {
+    i = i * 10000 + parse_four_digits_unrolled(p);
+    p += 4;
+  }
+#endif // SIMDJSON_SWAR_NUMBER_PARSING
+#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
+  while (parse_digit(*p, i)) { p++; }
+}

+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_float_integer_digits(const uint8_t *&p, uint64_t &i,
+                                                bool &leading_zero) {
+  if (!parse_digit(*p, i)) { leading_zero = true; return; }
+  p++;
+  leading_zero = (i == 0);
+  while (parse_digit(*p, i)) { p++; }
+}
+
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_integer_digits(const uint8_t *&p, uint64_t &i) {
 #ifdef SIMDJSON_SWAR_NUMBER_PARSING
 #if SIMDJSON_SWAR_NUMBER_PARSING
-  // this helps if we have lots of decimals!
-  // this turns out to be frequent enough.
+  // Identifiers, timestamps and counters often have eight digits or more.
+  // Prior related work: jsonifier parses integers as eight-digit SWAR words
+  // (str_to_i.hpp, https://github.com/nihilai-collective/Jsonifier). This
+  // takes one such word with parse_eight_digits_unrolled, the routine
+  // simdjson uses for long fractions.
   if (is_made_of_eight_digits_fast(p)) {
     i = i * 100000000 + parse_eight_digits_unrolled(p);
     p += 8;
   }
+  const uint8_t *const swar_end = p + 8;
+  while (p < swar_end && is_made_of_four_digits_fast(p)) {
+    i = i * 10000 + parse_four_digits_unrolled(p);
+    p += 4;
+  }
 #endif // SIMDJSON_SWAR_NUMBER_PARSING
 #endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
-  // Unrolling the first digit makes a small difference on some implementations (e.g. westmere)
-  if (parse_digit(*p, i)) { ++p; }
   while (parse_digit(*p, i)) { p++; }
+}
+
+simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
+  // we continue with the fiction that we have an integer. If the
+  // floating point number is representable as x * 10^z for some integer
+  // z that fits in 53 bits, then we will be able to convert back the
+  // the integer into a float in a lossless manner.
+  const uint8_t *const first_after_period = p;
+  parse_fraction_digits(p, i);
   exponent = first_after_period - p;
   // Decimal without digits (123.) is illegal
   if (exponent == 0) {
@@ -23678,7 +31593,7 @@ simdjson_inline size_t significant_digits(const uint8_t * start_digits, size_t d
 } // unnamed namespace

 /** @private */
-static error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
+inline error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
   if (parse_float_fallback(src, answer)) {
     return SUCCESS;
   }
@@ -23761,15 +31676,17 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
   return SUCCESS;              // always succeeds
 }

-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept  { return false; }
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept  { return false; }
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept { return number_type::signed_integer; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline bool is_negative(const uint8_t *) noexcept  { return false; }
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t *) noexcept  { return false; }
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t *) noexcept { return number_type::signed_integer; }
 #else

 // parse the number at src
@@ -23800,7 +31717,21 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
   size_t digit_count = size_t(p - start_digits);
-  if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) { return INVALID_NUMBER(src); }
+  if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // By this point, we know that our input does not begin with a digit. We will attempt
+    // to handle NaN/Infinity.
+
+    double d;
+    if (compute_nan_inf(p, negative, d)) {
+      writer.append_double(d);
+      return SUCCESS;
+    }
+#endif
+
+    return INVALID_NUMBER(src);
+  }

   //
   // Handle floats if there is a . or e (or both)
@@ -23849,7 +31780,7 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
     // - Therefore, if the number is positive and lower than that, it's overflow.
     // - The value we are looking at is less than or equal to INT64_MAX.
     //
-    }  else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return INVALID_NUMBER(src); }
+    }  else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return BIGINT_NUMBER(src); }
   }

   // Write unsigned if it does not fit in a signed integer.
@@ -23948,7 +31879,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const u
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -24046,7 +31977,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_stri
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -24101,7 +32032,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uin
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -24187,7 +32118,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = src;
   uint64_t i = 0;
-  while (parse_digit(*src, i)) { src++; }
+  parse_integer_digits(src, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -24227,9 +32158,247 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   //
   uint64_t i = 0;
   const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no loading digits, the number may be nan or infinity.
+    // Attempt to compute those, and return on success.
+    double d;
+    if (compute_nan_inf(p, negative, d)) { return d; }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = p-start_digits > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if (*p == 'e' || *p == 'E') {
+    p++;
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while (parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  double d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_64(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
+// Parse a JSON number into a binary32 (float) value.
+//
+// This mirrors parse_double, but it rounds to binary32 directly instead of
+// rounding to binary64 and then to binary32: the latter is subject to double
+// rounding and would not always produce the nearest float.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * src) noexcept {
+  //
+  // Check for minus sign
+  //
+  bool negative = (*src == '-');
+  src += uint8_t(negative);
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, the number may be nan or infinity.
+    // Attempt to compute those, and return on success.
+    float f;
+    if (compute_nan_inf(p, negative, f)) { return f; }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = p-start_digits > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if (*p == 'e' || *p == 'E') {
+    p++;
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while (parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  float d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_32(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
+simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
+  return (*src == '-');
+}
+
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept {
+  bool negative = (*src == '-');
+  src += uint8_t(negative);
+  const uint8_t *p = src;
+  while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
+  if ( p == src ) { return NUMBER_ERROR; }
+  if (jsoncharutils::is_structural_or_whitespace(*p)) { return true; }
+  return false;
+}
+
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept {
+  bool negative = (*src == '-');
+  src += uint8_t(negative);
+  const uint8_t *p = src;
+  while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
+  size_t digit_count = size_t(p - src);
+  if ( p == src ) { return NUMBER_ERROR; }
+  if (jsoncharutils::is_structural_or_whitespace(*p)) {
+    static const uint8_t * smaller_big_integer = reinterpret_cast<const uint8_t *>("9223372036854775808");
+    // We have an integer.
+    if(simdjson_unlikely(digit_count > 20)) {
+      return number_type::big_integer;
+    }
+    // If the number is negative and valid, it must be a signed integer.
+    if(negative) {
+      if (simdjson_unlikely(digit_count > 19)) return number_type::big_integer;
+      if (simdjson_unlikely(digit_count == 19 && memcmp(src, smaller_big_integer, 19) > 0)) {
+        return number_type::big_integer;
+      }
+#if SIMDJSON_MINUS_ZERO_AS_FLOAT
+      if(digit_count == 1 && src[0] == '0') {
+        // We have to write -0.0 instead of 0
+        return number_type::floating_point_number;
+      }
+#endif
+      return number_type::signed_integer;
+    }
+    // Let us check if we have a big integer (>=2**64).
+    static const uint8_t * two_to_sixtyfour = reinterpret_cast<const uint8_t *>("18446744073709551616");
+    if((digit_count > 20) || (digit_count == 20 && memcmp(src, two_to_sixtyfour, 20) >= 0)) {
+      return number_type::big_integer;
+    }
+    // The number is positive and smaller than 18446744073709551616 (or 2**64).
+    // We want values larger or equal to 9223372036854775808 to be unsigned
+    // integers, and the other values to be signed integers.
+    if((digit_count == 20) || (digit_count >= 19 && memcmp(src, smaller_big_integer, 19) >= 0)) {
+      return number_type::unsigned_integer;
+    }
+    return number_type::signed_integer;
+  }
+  // Hopefully, we have 'e' or 'E' or '.'.
+  return number_type::floating_point_number;
+}
+
+// Never read at src_end or beyond
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * src, const uint8_t * const src_end) noexcept {
+  if(src == src_end) { return NUMBER_ERROR; }
+  //
+  // Check for minus sign
+  //
+  bool negative = (*src == '-');
+  src += uint8_t(negative);
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  if(p == src_end) { return NUMBER_ERROR; }
   p += parse_digit(*p, i);
   bool leading_zero = (i == 0);
-  while (parse_digit(*p, i)) { p++; }
+  while ((p != src_end) && parse_digit(*p, i)) { p++; }
   // no integer digits, or 0123 (zero must be solo)
   if ( p == src ) { return INCORRECT_TYPE; }
   if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
@@ -24239,12 +32408,104 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   //
   int64_t exponent = 0;
   bool overflow;
-  if (simdjson_likely(*p == '.')) {
+  if (simdjson_likely((p != src_end) && (*p == '.'))) {
     p++;
     const uint8_t *start_decimal_digits = p;
-    if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
+    if ((p == src_end) || !parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
     p++;
-    while (parse_digit(*p, i)) { p++; }
+    while ((p != src_end) && parse_digit(*p, i)) { p++; }
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = start_digits-src > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if ((p != src_end) && (*p == 'e' || *p == 'E')) {
+    p++;
+    if(p == src_end) { return NUMBER_ERROR; }
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while ((p != src_end) && parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if ((p != src_end) && jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  double d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_64(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), src_end, &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * src) noexcept {
+  //
+  // Check for minus sign
+  //
+  bool negative = (*(src + 1) == '-');
+  src += uint8_t(negative) + 1;
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, attempt to parse numbers that are either
+    // NaN or Infinity
+    if (atomparsing::is_valid_inf_in_string(src)) {
+      double inf = std::numeric_limits<double>::infinity();
+      return negative ? -inf : inf;
+    }
+
+    if (atomparsing::is_valid_nan_in_string(src)) {
+      return std::numeric_limits<double>::quiet_NaN();
+    }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
     exponent = -(p - start_decimal_digits);

     // Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -24276,7 +32537,7 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
     exponent += exp_neg ? 0-exp : exp;
   }

-  if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+  if (*p != '"') { return NUMBER_ERROR; }

   overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;

@@ -24293,163 +32554,39 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   return d;
 }

-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
-  return (*src == '-');
-}
-
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept {
-  bool negative = (*src == '-');
-  src += uint8_t(negative);
-  const uint8_t *p = src;
-  while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
-  if ( p == src ) { return NUMBER_ERROR; }
-  if (jsoncharutils::is_structural_or_whitespace(*p)) { return true; }
-  return false;
-}
-
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept {
-  bool negative = (*src == '-');
-  src += uint8_t(negative);
-  const uint8_t *p = src;
-  while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
-  size_t digit_count = size_t(p - src);
-  if ( p == src ) { return NUMBER_ERROR; }
-  if (jsoncharutils::is_structural_or_whitespace(*p)) {
-    static const uint8_t * smaller_big_integer = reinterpret_cast<const uint8_t *>("9223372036854775808");
-    // We have an integer.
-    if(simdjson_unlikely(digit_count > 20)) {
-      return number_type::big_integer;
-    }
-    // If the number is negative and valid, it must be a signed integer.
-    if(negative) {
-      if (simdjson_unlikely(digit_count > 19)) return number_type::big_integer;
-      if (simdjson_unlikely(digit_count == 19 && memcmp(src, smaller_big_integer, 19) > 0)) {
-        return number_type::big_integer;
-      }
-#if SIMDJSON_MINUS_ZERO_AS_FLOAT
-      if(digit_count == 1 && src[0] == '0') {
-        // We have to write -0.0 instead of 0
-        return number_type::floating_point_number;
-      }
-#endif
-      return number_type::signed_integer;
-    }
-    // Let us check if we have a big integer (>=2**64).
-    static const uint8_t * two_to_sixtyfour = reinterpret_cast<const uint8_t *>("18446744073709551616");
-    if((digit_count > 20) || (digit_count == 20 && memcmp(src, two_to_sixtyfour, 20) >= 0)) {
-      return number_type::big_integer;
-    }
-    // The number is positive and smaller than 18446744073709551616 (or 2**64).
-    // We want values larger or equal to 9223372036854775808 to be unsigned
-    // integers, and the other values to be signed integers.
-    if((digit_count == 20) || (digit_count >= 19 && memcmp(src, smaller_big_integer, 19) >= 0)) {
-      return number_type::unsigned_integer;
-    }
-    return number_type::signed_integer;
-  }
-  // Hopefully, we have 'e' or 'E' or '.'.
-  return number_type::floating_point_number;
-}
-
-// Never read at src_end or beyond
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * src, const uint8_t * const src_end) noexcept {
-  if(src == src_end) { return NUMBER_ERROR; }
+// Parse a JSON number held inside a JSON string into a binary32 (float) value.
+// See parse_float for why we do not simply round parse_double_in_string.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * src) noexcept {
   //
   // Check for minus sign
   //
-  bool negative = (*src == '-');
-  src += uint8_t(negative);
+  bool negative = (*(src + 1) == '-');
+  src += uint8_t(negative) + 1;

   //
   // Parse the integer part.
   //
   uint64_t i = 0;
   const uint8_t *p = src;
-  if(p == src_end) { return NUMBER_ERROR; }
-  p += parse_digit(*p, i);
-  bool leading_zero = (i == 0);
-  while ((p != src_end) && parse_digit(*p, i)) { p++; }
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
   // no integer digits, or 0123 (zero must be solo)
-  if ( p == src ) { return INCORRECT_TYPE; }
-  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
-
-  //
-  // Parse the decimal part.
-  //
-  int64_t exponent = 0;
-  bool overflow;
-  if (simdjson_likely((p != src_end) && (*p == '.'))) {
-    p++;
-    const uint8_t *start_decimal_digits = p;
-    if ((p == src_end) || !parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
-    p++;
-    while ((p != src_end) && parse_digit(*p, i)) { p++; }
-    exponent = -(p - start_decimal_digits);
-
-    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
-    overflow = p-src-1 > 19;
-    if (simdjson_unlikely(overflow && leading_zero)) {
-      // Skip leading 0.00000 and see if it still overflows
-      const uint8_t *start_digits = src + 2;
-      while (*start_digits == '0') { start_digits++; }
-      overflow = start_digits-src > 19;
+  if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, attempt to parse numbers that are either
+    // NaN or Infinity
+    if (atomparsing::is_valid_inf_in_string(src)) {
+      float inf = std::numeric_limits<float>::infinity();
+      return negative ? -inf : inf;
     }
-  } else {
-    overflow = p-src > 19;
-  }
-
-  //
-  // Parse the exponent
-  //
-  if ((p != src_end) && (*p == 'e' || *p == 'E')) {
-    p++;
-    if(p == src_end) { return NUMBER_ERROR; }
-    bool exp_neg = *p == '-';
-    p += exp_neg || *p == '+';
-
-    uint64_t exp = 0;
-    const uint8_t *start_exp_digits = p;
-    while ((p != src_end) && parse_digit(*p, exp)) { p++; }
-    // no exp digits, or 20+ exp digits
-    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
-
-    exponent += exp_neg ? 0-exp : exp;
-  }
-
-  if ((p != src_end) && jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }

-  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+    if (atomparsing::is_valid_nan_in_string(src)) {
+      return std::numeric_limits<float>::quiet_NaN();
+    }
+#endif

-  //
-  // Assemble (or slow-parse) the float
-  //
-  double d;
-  if (simdjson_likely(!overflow)) {
-    if (compute_float_64(exponent, i, negative, d)) { return d; }
+    return INCORRECT_TYPE;
   }
-  if (!parse_float_fallback(src - uint8_t(negative), src_end, &d)) {
-    return NUMBER_ERROR;
-  }
-  return d;
-}
-
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * src) noexcept {
-  //
-  // Check for minus sign
-  //
-  bool negative = (*(src + 1) == '-');
-  src += uint8_t(negative) + 1;
-
-  //
-  // Parse the integer part.
-  //
-  uint64_t i = 0;
-  const uint8_t *p = src;
-  p += parse_digit(*p, i);
-  bool leading_zero = (i == 0);
-  while (parse_digit(*p, i)) { p++; }
-  // no integer digits, or 0123 (zero must be solo)
-  if ( p == src ) { return INCORRECT_TYPE; }
   if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }

   //
@@ -24460,9 +32597,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   if (simdjson_likely(*p == '.')) {
     p++;
     const uint8_t *start_decimal_digits = p;
-    if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
-    p++;
-    while (parse_digit(*p, i)) { p++; }
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
     exponent = -(p - start_decimal_digits);

     // Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -24501,9 +32637,9 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   //
   // Assemble (or slow-parse) the float
   //
-  double d;
+  float d;
   if (simdjson_likely(!overflow)) {
-    if (compute_float_64(exponent, i, negative, d)) { return d; }
+    if (compute_float_32(exponent, i, negative, d)) { return d; }
   }
   if (!parse_float_fallback(src - uint8_t(negative), &d)) {
     return NUMBER_ERROR;
@@ -24866,16 +33002,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
 }
 #endif

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
-                                uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
-  return _addcarry_u64(0, value1, value2,
-                       reinterpret_cast<unsigned __int64 *>(result));
-#else
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-#endif
-}

 } // unnamed namespace
 } // namespace icelake
@@ -26428,6 +34554,296 @@ simdjson_inline uint32_t find_next_document_index(dom_parser_implementation &par
   return 0;
 }

+/**
+ * Sentinel value returned to indicate a document started but didn't fit
+ * (CAPACITY error), as opposed to 0 which means no document content found
+ * (EMPTY).
+ */
+constexpr uint32_t DOCUMENT_TOO_LARGE = UINT32_MAX;
+
+/**
+ * For RFC 7464 JSON text sequences, filter RS from structural indexes and
+ * find batch boundaries.
+ *
+ * In JSON sequence mode, RS (0x1E) marks the start of each JSON text.
+ * RS bytes appear in structural_indexes as they are classified as scalars.
+ * This function:
+ * 1. Scans structural_indexes to find and count RS positions
+ * 2. Filters RS out of structural_indexes in-place
+ * 3. Determines batch boundaries based on RS positions
+ *
+ * @param parser The parser with structural_indexes and buf.
+ * @param len The length of the current batch buffer.
+ * @param is_final True if this is the final batch (no more data coming).
+ * @param next_batch_start Output: offset where the next batch should start.
+ * @param scan_len Offset past which no value start may be derived. Equals len
+ *        unless stage 1 dropped a trailing unclosed string, whose bytes it
+ *        never validated.
+ * @return The number of structural indexes to keep (after RS filtering),
+ *         0 if no document content found (EMPTY),
+ *         or DOCUMENT_TOO_LARGE if a document started but didn't fit (CAPACITY).
+ */
+simdjson_inline uint32_t find_next_document_index_json_sequence(
+    dom_parser_implementation &parser,
+    size_t len,
+    bool is_final,
+    uint32_t &next_batch_start,
+    size_t scan_len) {
+  // Default: next batch starts at end of buffer
+  next_batch_start = uint32_t(len);
+
+  if (parser.n_structural_indexes == 0) { return 0; }
+
+  // Phase 1: Scan structural_indexes to find RS positions and handle them.
+  // RS marks the start of a JSON text. For objects/arrays, the '{' or '[' after RS
+  // is already in structural_indexes (it's an operator). For scalars like numbers,
+  // the digit following RS is NOT in structural_indexes because the scanner sees
+  // RS as a scalar, making the digit a scalar continuation, not a start.
+  // We must: (1) remove RS from structural_indexes, and (2) for scalars, add the
+  // actual value start position.
+  // The EOF sentinel: len, or where a discarded unclosed string starts.
+  const uint32_t sentinel = parser.structural_indexes[parser.n_structural_indexes];
+  uint32_t write_idx = 0;
+  uint32_t last_rs_pos = 0;
+  uint32_t rs_count = 0;
+
+  for (uint32_t read_idx = 0; read_idx < parser.n_structural_indexes; read_idx++) {
+    const uint32_t pos = parser.structural_indexes[read_idx];
+    if (parser.buf[pos] == 0x1E) {
+      // This is an RS character - find the actual JSON value start.
+      last_rs_pos = pos;
+      rs_count++;
+      // Skip past this RS and any whitespace *and any additional RSes*
+      // to locate the real value. Consecutive RSes are degenerate
+      // "empty records" per RFC 7464; we collapse them here. They do
+      // not always appear as separate entries in structural_indexes
+      // because the scanner groups runs of adjacent non-whitespace
+      // scalar bytes (including RS) into a single scalar start.
+      uint32_t value_start = pos + 1;
+      while (value_start < scan_len) {
+        const uint8_t c = parser.buf[value_start];
+        if (c == ' ' || c == '\t' || c == '\n' || c == '\r') {
+          value_start++;
+        } else if (c == 0x1E) {
+          // Collapsed empty record. Still count it so rs_count reflects
+          // the true number of record markers and last_rs_pos tracks
+          // the final one.
+          last_rs_pos = value_start;
+          rs_count++;
+          value_start++;
+        } else {
+          break;
+        }
+      }
+      // If the scanner emitted additional structurals inside the
+      // whitespace+RS run we just walked over (i.e., isolated RSes
+      // separated by whitespace), skip past them so we do not
+      // double-count or double-emit.
+      while (read_idx + 1 < parser.n_structural_indexes &&
+             parser.structural_indexes[read_idx + 1] < value_start) {
+        read_idx++;
+      }
+      // Check if the value start is an operator (always present in
+      // scanner structural_indexes) or a scalar-like start (which may
+      // be missing from structural_indexes and must be added here).
+      // Note: '"' is NOT always in structural_indexes. The scanner
+      // classifies '"' as a scalar character and emits it as a
+      // structural only when it is a *scalar start* (preceded by
+      // whitespace or an operator). When '"' immediately follows an
+      // RS (which the scanner also classifies as scalar), it is
+      // treated as a scalar continuation and not emitted - so we
+      // must add it here just like any other scalar value.
+      if (value_start < scan_len) {
+        const uint8_t c = parser.buf[value_start];
+        const bool is_operator =
+            (c == '{' || c == '}' || c == '[' || c == ']' ||
+             c == ':' || c == ',');
+        // If the next scanner structural is exactly at value_start,
+        // the scanner already emitted it (it followed whitespace) and
+        // we must not add a duplicate - a subsequent iteration will
+        // copy it into write_idx.
+        const bool already_emitted =
+            (read_idx + 1 < parser.n_structural_indexes &&
+             parser.structural_indexes[read_idx + 1] == value_start);
+        if (!is_operator && !already_emitted) {
+          // Scalar value (number/true/false/null/string) - add its
+          // position since scanner missed it.
+          parser.structural_indexes[write_idx++] = value_start;
+        }
+      }
+    } else {
+      // Not RS, copy to output
+      parser.structural_indexes[write_idx++] = pos;
+    }
+  }
+
+  // Update structural index count. Compaction left a stale index in the slot
+  // past the end: restore the EOF sentinel that stage 1 had planted there, which
+  // document_stream::truncated_bytes() reads after a final batch.
+  parser.n_structural_indexes = write_idx;
+  parser.structural_indexes[write_idx] = sentinel;
+
+  if (parser.n_structural_indexes == 0) {
+    // Only RS markers here: the last one opens a record continuing past the
+    // window, so that is where the next batch resumes.
+    if (!is_final && rs_count > 0) { next_batch_start = last_rs_pos; }
+    return 0;
+  }
+  if (rs_count == 0) {
+    // No RS found; for final batch, try generic boundary detection
+    return is_final ? find_next_document_index(parser) : 0;
+  }
+
+  // Phase 2: Determine batch boundaries based on RS positions
+
+  if (is_final) {
+    // Final batch: all documents are complete (last one ends at EOF).
+    // In json_sequence mode, RS markers define document boundaries, so all
+    // remaining structurals form complete documents. Return them all directly.
+    // (Calling find_next_document_index() would fail for scalar documents.)
+    return parser.n_structural_indexes;
+  }
+
+  // Partial batch: need to find complete documents only.
+  // A document starting at an RS is complete if there is another RS after it.
+  next_batch_start = last_rs_pos;
+
+  if (rs_count < 2) {
+    // Only one RS, so we have at most one document that may be incomplete.
+    // We cannot confirm it is complete without another RS.
+    // Return DOCUMENT_TOO_LARGE if content was found (write_idx > 0), 0 if only separators.
+    return (parser.n_structural_indexes > 0) ? DOCUMENT_TOO_LARGE : 0;
+  }
+
+  // We have at least 2 RS markers. The last complete document ends before last_rs_pos.
+
+  // Find the structural index cutoff: keep only structurals < last_rs_pos.
+  // Since we already filtered RS, all remaining structurals are valid.
+  // We iterate backward to find the last structural before last_rs_pos.
+  uint32_t keep_count = 0;
+  for (uint32_t i = parser.n_structural_indexes; i > 0; i--) {
+    if (parser.structural_indexes[i - 1] < last_rs_pos) {
+      keep_count = i;
+      break;
+    }
+  }
+
+  // No structurals before the last RS - no complete documents
+  if (keep_count == 0) { return 0; }
+
+  // All documents before the last RS are complete by definition (the next RS
+  // confirms their end). No need to call find_next_document_index() which
+  // would fail for scalar documents like `1` or `"hello"`.
+  return keep_count;
+}
+
+/**
+ * Filter comma-delimited documents by removing root-level commas from
+ * structural indexes.
+ *
+ * For comma-delimited format like `{...},{...},{...}`, we need to remove
+ * the commas that separate documents (depth 0) while preserving commas
+ * inside arrays and objects (depth > 0).
+ *
+ * After filtering, the structural indexes look like whitespace-delimited
+ * documents, so find_next_document_index() works unchanged.
+ *
+ * @param parser The parser with structural_indexes and buf.
+ * @param len The length of the current batch buffer.
+ * @param is_final True if this is the final batch (no more data coming).
+ * @param next_batch_start Output: offset where the next batch should start.
+ * @return The number of structural indexes to keep,
+ *         0 if no document content found (EMPTY),
+ *         or DOCUMENT_TOO_LARGE if a document started but didn't fit (CAPACITY).
+ */
+simdjson_inline uint32_t filter_comma_delimited(
+    dom_parser_implementation &parser,
+    size_t len,
+    bool is_final,
+    uint32_t &next_batch_start) {
+  // Default: next batch starts at end of buffer
+  next_batch_start = uint32_t(len);
+
+  if (parser.n_structural_indexes == 0) { return 0; }
+
+  // Track depth to identify root-level commas (depth 0)
+  int depth = 0;
+  // The EOF sentinel: len, or where a discarded unclosed string starts.
+  const uint32_t sentinel = parser.structural_indexes[parser.n_structural_indexes];
+  uint32_t write_idx = 0;
+  uint32_t last_root_comma_pos = 0;
+  uint32_t root_comma_count = 0;
+
+  for (uint32_t i = 0; i < parser.n_structural_indexes; i++) {
+    uint32_t idx = parser.structural_indexes[i];
+    uint8_t c = parser.buf[idx];
+
+    switch (c) {
+      case '{': case '[':
+        depth++;
+        parser.structural_indexes[write_idx++] = idx;
+        break;
+      case '}': case ']':
+        depth--;
+        parser.structural_indexes[write_idx++] = idx;
+        break;
+      case ',':
+        if (depth == 0) {
+          // Root-level comma = document boundary, skip it
+          last_root_comma_pos = idx;
+          root_comma_count++;
+          continue;
+        }
+        parser.structural_indexes[write_idx++] = idx;
+        break;
+      default:
+        // Colons, scalars, etc.
+        parser.structural_indexes[write_idx++] = idx;
+        break;
+    }
+  }
+
+  // Update structural index count. Compaction left a stale index in the slot
+  // past the end: restore the EOF sentinel that stage 1 had planted there, which
+  // document_stream::truncated_bytes() reads after a final batch.
+  parser.n_structural_indexes = write_idx;
+  parser.structural_indexes[write_idx] = sentinel;
+
+  if (parser.n_structural_indexes == 0) { return 0; }
+
+  if (is_final) {
+    // Final batch: use standard boundary detection on filtered indexes
+    return find_next_document_index(parser);
+  }
+
+  // Partial batch: need to find complete documents only.
+  // A document ending with a root comma is complete.
+  if (root_comma_count == 0) {
+    // No root commas found; we cannot confirm any document is complete.
+    // The whole batch might be one incomplete document.
+    // Return DOCUMENT_TOO_LARGE if content was found (write_idx > 0), 0 if only commas.
+    return (parser.n_structural_indexes > 0) ? DOCUMENT_TOO_LARGE : 0;
+  }
+
+  // We have at least one root comma. Documents before the last comma are complete.
+  next_batch_start = last_root_comma_pos + 1;
+
+  // Find the structural index cutoff: keep only structurals < last_root_comma_pos
+  uint32_t keep_count = 0;
+  for (uint32_t i = parser.n_structural_indexes; i > 0; i--) {
+    if (parser.structural_indexes[i - 1] < last_root_comma_pos) {
+      keep_count = i;
+      break;
+    }
+  }
+
+  if (keep_count == 0) { return 0; }
+
+  // Use standard boundary detection on the complete portion
+  parser.n_structural_indexes = keep_count;
+  return find_next_document_index(parser);
+}
+
 } // namespace stage1
 } // unnamed namespace
 } // namespace icelake
@@ -26861,7 +35277,6 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
         return EMPTY;
       }
     }
-
     parser.n_structural_indexes = new_structural_indexes;
   } else if (partial == stage1_mode::streaming_final) {
     if(have_unclosed_string) { parser.n_structural_indexes--; }
@@ -26889,6 +35304,75 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
         // the trailing garbage.
         return EMPTY;
     }
+  } else if (partial == stage1_mode::json_sequence_partial) {
+    // RFC 7464: use RS positions for batch boundaries
+    // A discarded unclosed string also caps how far the filter may scan.
+    size_t scan_len = len;
+    if(have_unclosed_string) {
+      parser.n_structural_indexes--;
+      if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return CAPACITY; }
+      scan_len = parser.structural_indexes[parser.n_structural_indexes];
+    }
+    uint32_t next_batch_start = uint32_t(len);
+    auto new_structural_indexes = find_next_document_index_json_sequence(parser, len, false, next_batch_start, scan_len);
+    if (new_structural_indexes == DOCUMENT_TOO_LARGE) {
+      return CAPACITY;
+    }
+    if (new_structural_indexes == 0) {
+      // An EMPTY batch must still advance next_batch_start, or document_stream
+      // re-parses the same bytes forever. CAPACITY when it cannot.
+      if (next_batch_start == 0) { return CAPACITY; }
+      parser.n_structural_indexes = 0;
+      parser.structural_indexes[0] = next_batch_start;
+      return EMPTY;
+    }
+    parser.n_structural_indexes = new_structural_indexes;
+    parser.structural_indexes[parser.n_structural_indexes] = next_batch_start;
+  } else if (partial == stage1_mode::json_sequence_final) {
+    // RFC 7464: final batch, last document extends to EOF
+    // As above: a discarded unclosed string caps how far the filter may scan.
+    size_t scan_len = len;
+    if(have_unclosed_string) {
+      parser.n_structural_indexes--;
+      scan_len = parser.structural_indexes[parser.n_structural_indexes];
+    }
+    uint32_t next_batch_start = uint32_t(len);
+    parser.n_structural_indexes = find_next_document_index_json_sequence(parser, len, true, next_batch_start, scan_len);
+    // The filter compacted structural_indexes in place and restored the EOF
+    // sentinel past the compacted end, so the copy below is either the start
+    // of a truncated document or len, as in streaming_final.
+    parser.structural_indexes[parser.n_structural_indexes + 1] = parser.structural_indexes[parser.n_structural_indexes];
+    parser.structural_indexes[parser.n_structural_indexes] = uint32_t(len);
+    if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return EMPTY; }
+  } else if (partial == stage1_mode::comma_delimited_partial) {
+    // Comma-delimited: filter root-level commas, use comma positions for batch boundaries
+    if(have_unclosed_string) {
+      parser.n_structural_indexes--;
+      if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return CAPACITY; }
+    }
+    uint32_t next_batch_start = uint32_t(len);
+    auto new_structural_indexes = filter_comma_delimited(parser, len, false, next_batch_start);
+    if (new_structural_indexes == DOCUMENT_TOO_LARGE) {
+      return CAPACITY;
+    }
+    if (new_structural_indexes == 0) {
+      // An EMPTY batch must still advance next_batch_start, or document_stream
+      // re-parses the same bytes forever. CAPACITY when it cannot.
+      if (next_batch_start == 0) { return CAPACITY; }
+      parser.n_structural_indexes = 0;
+      parser.structural_indexes[0] = next_batch_start;
+      return EMPTY;
+    }
+    parser.n_structural_indexes = new_structural_indexes;
+    parser.structural_indexes[parser.n_structural_indexes] = next_batch_start;
+  } else if (partial == stage1_mode::comma_delimited_final) {
+    // Comma-delimited: final batch, last document extends to EOF
+    if(have_unclosed_string) { parser.n_structural_indexes--; }
+    uint32_t next_batch_start = uint32_t(len);
+    parser.n_structural_indexes = filter_comma_delimited(parser, len, true, next_batch_start);
+    parser.structural_indexes[parser.n_structural_indexes + 1] = parser.structural_indexes[parser.n_structural_indexes];
+    parser.structural_indexes[parser.n_structural_indexes] = uint32_t(len);
+    if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return EMPTY; }
   }
   checker.check_eof();
   return checker.errors();
@@ -26971,7 +35455,6 @@ namespace {
 namespace stage2 {

 class json_iterator;
-class structural_iterator;
 struct tape_builder;
 struct tape_writer;

@@ -27268,7 +35751,7 @@ public:
    *
    * - increment_count(iter) - each time a value is found in an array or object.
    */
-  template<bool STREAMING, typename V>
+  template<bool STREAMING, bool UNPADDED, typename V>
   simdjson_warn_unused simdjson_inline error_code walk_document(V &visitor) noexcept;

   /**
@@ -27285,6 +35768,7 @@ public:
    *
    * They may include invalid JSON as well (such as `1.2.3` or `ture`).
    */
+  template<bool UNPADDED>
   simdjson_inline const uint8_t *peek() const noexcept;
   /**
    * Advance to the next token.
@@ -27293,6 +35777,7 @@ public:
    *
    * They may include invalid JSON as well (such as `1.2.3` or `ture`).
    */
+  template<bool UNPADDED>
   simdjson_inline const uint8_t *advance() noexcept;
   /**
    * Get the remaining length of the document, from the start of the current token.
@@ -27341,7 +35826,7 @@ public:
   simdjson_warn_unused simdjson_inline error_code visit_primitive(V &visitor, const uint8_t *value) noexcept;
 };

-template<bool STREAMING, typename V>
+template<bool STREAMING, bool UNPADDED, typename V>
 simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &visitor) noexcept {
   logger::log_start();

@@ -27356,7 +35841,7 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &
   // Read first value
   //
   {
-    auto value = advance();
+    auto value = advance<UNPADDED>();

     // Make sure the outer object or array is closed before continuing; otherwise, there are ways we
     // could get into memory corruption. See https://github.com/simdjson/simdjson/issues/906
@@ -27368,8 +35853,8 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &
     }

     switch (*value) {
-      case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
-      case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+      case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+      case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
       default: SIMDJSON_TRY( visitor.visit_root_primitive(*this, value) ); break;
     }
   }
@@ -27386,29 +35871,29 @@ object_begin:
   SIMDJSON_TRY( visitor.visit_object_start(*this) );

   {
-    auto key = advance();
+    auto key = advance<UNPADDED>();
     if (*key != '"') { log_error("Object does not start with a key"); return TAPE_ERROR; }
     SIMDJSON_TRY( visitor.increment_count(*this) );
     SIMDJSON_TRY( visitor.visit_key(*this, key) );
   }

 object_field:
-  if (simdjson_unlikely( *advance() != ':' )) { log_error("Missing colon after key in object"); return TAPE_ERROR; }
+  if (simdjson_unlikely( *advance<UNPADDED>() != ':' )) { log_error("Missing colon after key in object"); return TAPE_ERROR; }
   {
-    auto value = advance();
+    auto value = advance<UNPADDED>();
     switch (*value) {
-      case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
-      case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+      case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+      case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
       default: SIMDJSON_TRY( visitor.visit_primitive(*this, value) ); break;
     }
   }

 object_continue:
-  switch (*advance()) {
+  switch (*advance<UNPADDED>()) {
     case ',':
       SIMDJSON_TRY( visitor.increment_count(*this) );
       {
-        auto key = advance();
+        auto key = advance<UNPADDED>();
         if (simdjson_unlikely( *key != '"' )) { log_error("Key string missing at beginning of field in object"); return TAPE_ERROR; }
         SIMDJSON_TRY( visitor.visit_key(*this, key) );
       }
@@ -27436,16 +35921,16 @@ array_begin:

 array_value:
   {
-    auto value = advance();
+    auto value = advance<UNPADDED>();
     switch (*value) {
-      case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
-      case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+      case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+      case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
       default: SIMDJSON_TRY( visitor.visit_primitive(*this, value) ); break;
     }
   }

 array_continue:
-  switch (*advance()) {
+  switch (*advance<UNPADDED>()) {
     case ',': SIMDJSON_TRY( visitor.increment_count(*this) ); goto array_value;
     case ']': log_end_value("array"); SIMDJSON_TRY( visitor.visit_array_end(*this) ); goto scope_end;
     default: log_error("Missing comma between array values"); return TAPE_ERROR;
@@ -27473,11 +35958,29 @@ simdjson_inline json_iterator::json_iterator(dom_parser_implementation &_dom_par
     dom_parser{_dom_parser} {
 }

+// Stage 1 leaves a sentinel index of value `len`; reading buf[len] requires
+// padding. For UNPADDED, return a dummy so walk_document can fail cleanly (#2815).
+template<bool UNPADDED>
 simdjson_inline const uint8_t *json_iterator::peek() const noexcept {
-  return &buf[*(next_structural)];
+  const uint32_t idx = *(next_structural);
+  SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+    if (simdjson_unlikely(idx >= dom_parser.len)) {
+      static constexpr uint8_t unpadded_eof_sentinel = 0;
+      return &unpadded_eof_sentinel;
+    }
+  }
+  return &buf[idx];
 }
+template<bool UNPADDED>
 simdjson_inline const uint8_t *json_iterator::advance() noexcept {
-  return &buf[*(next_structural++)];
+  const uint32_t idx = *(next_structural++);
+  SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+    if (simdjson_unlikely(idx >= dom_parser.len)) {
+      static constexpr uint8_t unpadded_eof_sentinel = 0;
+      return &unpadded_eof_sentinel;
+    }
+  }
+  return &buf[idx];
 }
 simdjson_inline size_t json_iterator::remaining_len() const noexcept {
   return dom_parser.len - *(next_structural-1);
@@ -27517,7 +36020,20 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::visit_root_primit
     case '"': return visitor.visit_root_string(*this, value);
     case 't': return visitor.visit_root_true_atom(*this, value);
     case 'f': return visitor.visit_root_false_atom(*this, value);
+#if SIMDJSON_ENABLE_NAN_INF
+    case 'n': {
+      auto err = visitor.visit_root_null_atom(*this, value);
+      if (err == SUCCESS) { return err; }
+      // propagate the error value returned by a bad 'null' atom if parsing 'nan' fails
+      return visitor.visit_root_nan_atom(*this, value, err);
+    }
+    // 'N' isn't a canonically recognized atom, so we return a TAPE_ERROR if failure occurs
+    case 'N': return visitor.visit_root_nan_atom(*this, value, TAPE_ERROR);
+    case 'i':
+    case 'I': return visitor.visit_root_inf_atom(*this, value);
+#else
     case 'n': return visitor.visit_root_null_atom(*this, value);
+#endif
     case '-':
     case '0': case '1': case '2': case '3': case '4':
     case '5': case '6': case '7': case '8': case '9':
@@ -27539,7 +36055,20 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::visit_primitive(V
   switch (*value) {
     case 't': return visitor.visit_true_atom(*this, value);
     case 'f': return visitor.visit_false_atom(*this, value);
+#if SIMDJSON_ENABLE_NAN_INF
+    case 'n': {
+      auto err = visitor.visit_null_atom(*this, value);
+      if (err == SUCCESS) { return err; }
+      // propagate the error value returned by a bad 'null' atom if parsing 'nan' fails
+      return visitor.visit_nan_atom(*this, value, err);
+    }
+    // 'N' isn't a canonically recognized atom, so we return a TAPE_ERROR if failure occurs
+    case 'N': return visitor.visit_nan_atom(*this, value, TAPE_ERROR);
+    case 'i':
+    case 'I': return visitor.visit_inf_atom(*this, value);
+#else
     case 'n': return visitor.visit_null_atom(*this, value);
+#endif
     default:
       log_error("Non-value found when value was expected!");
       return TAPE_ERROR;
@@ -27725,9 +36254,13 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_string(const uint8_t *src, u
            within the unicode codepoint handling code. */
         src += bs_dist;
         dst += bs_dist;
-        if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
-          return nullptr;
-        }
+        // Decode adjacent Unicode escapes without returning to the
+        // quote-and-backslash scanner between code points.
+        do {
+          if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
+            return nullptr;
+          }
+        } while (src[0] == '\\' && src[1] == 'u');
       } else {
         /* simple 1:1 conversion. Will eat bs_dist+2 characters in input and
          * write bs_dist+1 characters to output
@@ -27750,6 +36283,77 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_string(const uint8_t *src, u
   }
 }

+/**
+ * Bounds-safe variant of parse_string for input buffers that are NOT padded to
+ * len + SIMDJSON_PADDING bytes. `buf_end` is one past the last readable input
+ * byte (buf + len). It behaves exactly like parse_string while we are at least
+ * SIMDJSON_PADDING bytes away from buf_end (so every speculative SIMD read stays
+ * in bounds); once within the final SIMDJSON_PADDING bytes it copies the few
+ * remaining bytes into a space-padded scratch buffer and finishes with the
+ * regular parse_string. This keeps the delicate escape/Unicode handling in one
+ * place (the proven parse_string) rather than duplicating it.
+ *
+ * Correctness relies on stage 1 having validated the string, i.e. there is an
+ * unescaped closing quote within [src, buf_end); that quote is therefore inside
+ * the copied scratch, so parse_string finds it without running off the scratch.
+ */
+simdjson_warn_unused simdjson_inline uint8_t *parse_string_safe(const uint8_t *src, uint8_t *dst, bool allow_replacement, const uint8_t *buf_end) {
+  // Far from the end: identical to parse_string's loop. The guard uses
+  // SIMDJSON_PADDING (>= BYTES_PROCESSED) so copy_and_find never reads past
+  // buf_end; escape/Unicode look-aheads read within the string (before the
+  // closing quote, which is < buf_end), so they are in bounds here too.
+  // We add margin (+12) for handle_unicode_codepoint's worst-case lookahead:
+  // after seeing a high surrogate, it does hex_to_u32_nocheck on the immediate
+  // following bytes (+6 from the '\'), then (if it sees \u) another
+  // hex_to_u32_nocheck at +8..+11 relative to the backslash that started the
+  // escape. With bs_dist up to BYTES_PROCESSED-1 this reaches +11 from the
+  // chunk start. The +12 margin ensures that even on kernels where
+  // BYTES_PROCESSED == SIMDJSON_PADDING (e.g. icelake) the 4-byte read stays
+  // in-bounds. The scratch fallback (3*PAD) is already safe.
+  while (src + SIMDJSON_PADDING + 12 <= buf_end) {
+    auto b = backslash_and_quote{};
+    auto bs_quote = b.copy_and_find(src, dst);
+    if (bs_quote.has_quote_first()) {
+      return dst + bs_quote.quote_index();
+    }
+    if (bs_quote.has_backslash()) {
+      auto bs_dist = bs_quote.backslash_index();
+      uint8_t escape_char = src[bs_dist + 1];
+      if (escape_char == 'u') {
+        src += bs_dist;
+        dst += bs_dist;
+        if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
+          return nullptr;
+        }
+      } else {
+        uint8_t escape_result = escape_map[escape_char];
+        if (escape_result == 0u) {
+          return nullptr;
+        }
+        dst[bs_dist] = escape_result;
+        src += bs_dist + 2;
+        dst += bs_dist + 1;
+      }
+    } else {
+      src += backslash_and_quote::BYTES_PROCESSED;
+      dst += backslash_and_quote::BYTES_PROCESSED;
+    }
+  }
+  // Within the final SIMDJSON_PADDING bytes: copy what remains into a
+  // space-padded scratch (spaces are neither quote nor backslash, so they do not
+  // disturb matching) and let the regular parser finish from there. The closing
+  // quote is within `remaining` (< SIMDJSON_PADDING), so parse_string finds it in
+  // the chunk starting at some offset <= remaining and reads at most
+  // BYTES_PROCESSED (<= SIMDJSON_PADDING) further -- i.e. under 2*SIMDJSON_PADDING.
+  // We size at 3x for a comfortable margin (the unicode look-ahead reads a few
+  // extra bytes past an escape).
+  uint8_t scratch[SIMDJSON_PADDING * 3];
+  const size_t remaining = size_t(buf_end - src); // < SIMDJSON_PADDING
+  std::memset(scratch, ' ', sizeof(scratch));
+  std::memcpy(scratch, src, remaining);
+  return parse_string(scratch, dst, allow_replacement);
+}
+
 simdjson_warn_unused simdjson_inline uint8_t *parse_wobbly_string(const uint8_t *src, uint8_t *dst) {
   // It is not ideal that this function is nearly identical to parse_string.
   while (1) {
@@ -27804,73 +36408,6 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_wobbly_string(const uint8_t

 #endif // SIMDJSON_SRC_GENERIC_STAGE2_STRINGPARSING_H
 /* end file generic/stage2/stringparsing.h for icelake */
-/* including generic/stage2/structural_iterator.h for icelake: #include <generic/stage2/structural_iterator.h> */
-/* begin file generic/stage2/structural_iterator.h for icelake */
-#ifndef SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H
-
-/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
-/* amalgamation skipped (editor-only): #define SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H */
-/* amalgamation skipped (editor-only): #include <generic/stage2/base.h> */
-/* amalgamation skipped (editor-only): #include <simdjson/generic/dom_parser_implementation.h> */
-/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
-
-namespace simdjson {
-namespace icelake {
-namespace {
-namespace stage2 {
-
-class structural_iterator {
-public:
-  const uint8_t* const buf;
-  uint32_t *next_structural;
-  dom_parser_implementation &dom_parser;
-
-  // Start a structural
-  simdjson_inline structural_iterator(dom_parser_implementation &_dom_parser, size_t start_structural_index)
-    : buf{_dom_parser.buf},
-      next_structural{&_dom_parser.structural_indexes[start_structural_index]},
-      dom_parser{_dom_parser} {
-  }
-  // Get the buffer position of the current structural character
-  simdjson_inline const uint8_t* current() {
-    return &buf[*(next_structural-1)];
-  }
-  // Get the current structural character
-  simdjson_inline char current_char() {
-    return buf[*(next_structural-1)];
-  }
-  // Get the next structural character without advancing
-  simdjson_inline char peek_next_char() {
-    return buf[*next_structural];
-  }
-  simdjson_inline const uint8_t* peek() {
-    return &buf[*next_structural];
-  }
-  simdjson_inline const uint8_t* advance() {
-    return &buf[*(next_structural++)];
-  }
-  simdjson_inline char advance_char() {
-    return buf[*(next_structural++)];
-  }
-  simdjson_inline size_t remaining_len() {
-    return dom_parser.len - *(next_structural-1);
-  }
-
-  simdjson_inline bool at_end() {
-    return next_structural == &dom_parser.structural_indexes[dom_parser.n_structural_indexes];
-  }
-  simdjson_inline bool at_beginning() {
-    return next_structural == dom_parser.structural_indexes.get();
-  }
-};
-
-} // namespace stage2
-} // unnamed namespace
-} // namespace icelake
-} // namespace simdjson
-
-#endif // SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H
-/* end file generic/stage2/structural_iterator.h for icelake */
 /* including generic/stage2/tape_builder.h for icelake: #include <generic/stage2/tape_builder.h> */
 /* begin file generic/stage2/tape_builder.h for icelake */
 #ifndef SIMDJSON_SRC_GENERIC_STAGE2_TAPE_BUILDER_H
@@ -27893,12 +36430,8 @@ namespace icelake {
 namespace {
 namespace stage2 {

-struct tape_builder {
-  template<bool STREAMING>
-  simdjson_warn_unused static simdjson_inline error_code parse_document(
-    dom_parser_implementation &dom_parser,
-    dom::document &doc) noexcept;
-
+template <bool UNPADDED>
+struct tape_builder_impl {
   /** Called when a non-empty document starts. */
   simdjson_warn_unused simdjson_inline error_code visit_document_start(json_iterator &iter) noexcept;
   /** Called when a non-empty document ends without error. */
@@ -27951,88 +36484,130 @@ struct tape_builder {
   simdjson_warn_unused simdjson_inline error_code visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept;
   simdjson_warn_unused simdjson_inline error_code visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept;

+#if SIMDJSON_ENABLE_NAN_INF
+  simdjson_warn_unused simdjson_inline error_code visit_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept;
+  simdjson_warn_unused simdjson_inline error_code visit_root_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept;
+  // Attempts to parse 'inf' or 'infinity' (case insensitive). Because neither are canonical atoms,
+  // this returns a tape error on failure.
+  simdjson_warn_unused simdjson_inline error_code visit_inf_atom(json_iterator &iter, const uint8_t *value) noexcept;
+  simdjson_warn_unused simdjson_inline error_code visit_root_inf_atom(json_iterator &iter, const uint8_t *value) noexcept;
+#endif
+
   /** Called each time a new field or element in an array or object is found. */
   simdjson_warn_unused simdjson_inline error_code increment_count(json_iterator &iter) noexcept;

   /** Next location to write to tape */
   tape_writer tape;
+public:
+  simdjson_inline tape_builder_impl(dom::document &doc) noexcept;
 private:
   /** Next write location in the string buf for stage 2 parsing */
   uint8_t *current_string_buf_loc;

-  simdjson_inline tape_builder(dom::document &doc) noexcept;
-
   simdjson_inline uint32_t next_tape_index(json_iterator &iter) const noexcept;
   simdjson_inline void start_container(json_iterator &iter) noexcept;
   simdjson_warn_unused simdjson_inline error_code end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept;
   simdjson_warn_unused simdjson_inline error_code empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept;
   simdjson_inline uint8_t *on_start_string(json_iterator &iter) noexcept;
   simdjson_inline void on_end_string(uint8_t *dst) noexcept;
-}; // struct tape_builder
+}; // struct tape_builder_impl

-template<bool STREAMING>
-simdjson_warn_unused simdjson_inline error_code tape_builder::parse_document(
-    dom_parser_implementation &dom_parser,
-    dom::document &doc) noexcept {
-  dom_parser.doc = &doc;
-  json_iterator iter(dom_parser, STREAMING ? dom_parser.next_structural_index : 0);
-  tape_builder builder(doc);
-  return iter.walk_document<STREAMING>(builder);
-}
+// Thin, non-templated entry so each architecture's stage2() keeps calling
+// tape_builder::parse_document<STREAMING> unchanged. It chooses the bounds-safe
+// (unpadded) or the regular (padded) tape_builder_impl ONCE per document, so the
+// choice is a compile-time constant inside the walk: the padded path carries no
+// extra branch or load (see tape_builder_impl::visit_string).
+struct tape_builder {
+  template<bool STREAMING>
+  simdjson_warn_unused static simdjson_inline error_code parse_document(
+      dom_parser_implementation &dom_parser, dom::document &doc) noexcept {
+    dom_parser.doc = &doc;
+    json_iterator iter(dom_parser, STREAMING ? dom_parser.next_structural_index : 0);
+    if (dom_parser._unpadded) {
+      tape_builder_impl<true> builder(doc);
+      return iter.walk_document<STREAMING, true>(builder);
+    } else {
+      tape_builder_impl<false> builder(doc);
+      return iter.walk_document<STREAMING, false>(builder);
+    }
+  }
+};

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_primitive(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_primitive(json_iterator &iter, const uint8_t *value) noexcept {
   return iter.visit_root_primitive(*this, value);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_primitive(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_primitive(json_iterator &iter, const uint8_t *value) noexcept {
   return iter.visit_primitive(*this, value);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_empty_object(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_empty_object(json_iterator &iter) noexcept {
   return empty_container(iter, internal::tape_type::START_OBJECT, internal::tape_type::END_OBJECT);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_empty_array(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_empty_array(json_iterator &iter) noexcept {
   return empty_container(iter, internal::tape_type::START_ARRAY, internal::tape_type::END_ARRAY);
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_document_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_document_start(json_iterator &iter) noexcept {
   start_container(iter);
   return SUCCESS;
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_object_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_object_start(json_iterator &iter) noexcept {
   start_container(iter);
   return SUCCESS;
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_array_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_array_start(json_iterator &iter) noexcept {
   start_container(iter);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_object_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_object_end(json_iterator &iter) noexcept {
   return end_container(iter, internal::tape_type::START_OBJECT, internal::tape_type::END_OBJECT);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_array_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_array_end(json_iterator &iter) noexcept {
   return end_container(iter, internal::tape_type::START_ARRAY, internal::tape_type::END_ARRAY);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_document_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_document_end(json_iterator &iter) noexcept {
   constexpr uint32_t start_tape_index = 0;
   tape.append(start_tape_index, internal::tape_type::ROOT);
   tape_writer::write(iter.dom_parser.doc->tape[start_tape_index], next_tape_index(iter), internal::tape_type::ROOT);
   return SUCCESS;
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_key(json_iterator &iter, const uint8_t *key) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_key(json_iterator &iter, const uint8_t *key) noexcept {
   return visit_string(iter, key, true);
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::increment_count(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::increment_count(json_iterator &iter) noexcept {
   iter.dom_parser.open_containers[iter.depth].count++; // we have a key value pair in the object at parser.dom_parser.depth - 1
   return SUCCESS;
 }

-simdjson_inline tape_builder::tape_builder(dom::document &doc) noexcept : tape{doc.tape.get()}, current_string_buf_loc{doc.string_buf.get()} {}
+template <bool UNPADDED>
+simdjson_inline tape_builder_impl<UNPADDED>::tape_builder_impl(dom::document &doc) noexcept : tape{doc.tape.get()}, current_string_buf_loc{doc.string_buf.get()} {}

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_string(json_iterator &iter, const uint8_t *value, bool key) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_string(json_iterator &iter, const uint8_t *value, bool key) noexcept {
   iter.log_value(key ? "key" : "string");
   uint8_t *dst = on_start_string(iter);
-  dst = stringparsing::parse_string(value+1, dst, false); // We do not allow replacement when the escape characters are invalid.
+  // We do not allow replacement when the escape characters are invalid.
+  // UNPADDED is a compile-time constant chosen once per document by
+  // tape_builder::parse_document, so the padded build instantiates only the
+  // plain parse_string call below -- no runtime branch and no flag load.
+  SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+    dst = stringparsing::parse_string_safe(value+1, dst, false, iter.buf + iter.dom_parser.len);
+  } else {
+    dst = stringparsing::parse_string(value+1, dst, false);
+  }
   if (dst == nullptr) {
     iter.log_error("Invalid escape in string");
     return STRING_ERROR;
@@ -28041,27 +36616,48 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_string(json_
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_string(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_string(json_iterator &iter, const uint8_t *value) noexcept {
   return visit_string(iter, value);
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_number(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_number(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("number");
-  error_code err = numberparsing::parse_number(value, tape);
+  const uint8_t *num = value;
+  std::unique_ptr<uint8_t[]> copy{}; // keeps a padded copy of the tail alive when used
+  SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+    // numberparsing reads ahead in 8-byte blocks for floats
+    // (is_made_of_eight_digits_fast reads up to 7 bytes past the digits), so a
+    // number whose digits reach the final bytes of an unpadded buffer would read
+    // past it. *(next_structural) is the offset of the token following this
+    // number, hence an upper bound on where the digits end; when that is within
+    // SIMDJSON_PADDING of the end we parse from a space-padded copy of the tail
+    // (mirroring visit_root_number). This fires only for numbers near the end.
+    if (simdjson_unlikely(*(iter.next_structural) + SIMDJSON_PADDING > iter.dom_parser.len)) {
+      const size_t rl = iter.remaining_len(); // bytes from `value` to the end of the document
+      copy.reset(new (std::nothrow) uint8_t[rl + SIMDJSON_PADDING]);
+      if (copy.get() == nullptr) { return MEMALLOC; }
+      std::memcpy(copy.get(), value, rl);
+      std::memset(copy.get() + rl, ' ', SIMDJSON_PADDING);
+      num = copy.get();
+    }
+  }
+  error_code err = numberparsing::parse_number(num, tape);
   if (simdjson_unlikely(err == BIGINT_ERROR &&
       iter.dom_parser._number_as_string)) {
     // Write big integer to string buffer using the same format as strings.
     // Scan digits the same way parse_number does (skip optional '-', then digits).
-    const uint8_t *p = value;
+    const uint8_t *p = num;
     if (*p == '-') p++;
     while (numberparsing::is_digit(*p)) p++;
     // The digit run must be terminated by a structural or whitespace character; otherwise the
     // token is malformed (e.g. "123456789123456789123x").
     if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
-    size_t len = size_t(p - value);
+    size_t len = size_t(p - num);
     tape.append(current_string_buf_loc - iter.dom_parser.doc->string_buf.get(), internal::tape_type::BIGINT);
     uint8_t *dst = current_string_buf_loc + sizeof(uint32_t);
-    memcpy(dst, value, len);
+    memcpy(dst, num, len);
     dst += len;
     on_end_string(dst);
     return SUCCESS;
@@ -28069,7 +36665,8 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_number(json_
   return err;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_number(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_number(json_iterator &iter, const uint8_t *value) noexcept {
   //
   // We need to make a copy to make sure that the string is space terminated.
   // This is not about padding the input, which should already padded up
@@ -28083,76 +36680,142 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_number(
   // practice unless you are in the strange scenario where you have many JSON
   // documents made of single atoms.
   //
-  std::unique_ptr<uint8_t[]>copy(new (std::nothrow) uint8_t[iter.remaining_len() + SIMDJSON_PADDING]);
+  // In a stream, the input goes on with other documents: copy up to the next
+  // structural only, not to the end of the batch.
+  const size_t len = (std::min)(iter.remaining_len(), size_t(*iter.next_structural) - size_t(*(iter.next_structural - 1)));
+  std::unique_ptr<uint8_t[]>copy(new (std::nothrow) uint8_t[len + SIMDJSON_PADDING]);
   if (copy.get() == nullptr) { return MEMALLOC; }
-  std::memcpy(copy.get(), value, iter.remaining_len());
-  std::memset(copy.get() + iter.remaining_len(), ' ', SIMDJSON_PADDING);
+  std::memcpy(copy.get(), value, len);
+  std::memset(copy.get() + len, ' ', SIMDJSON_PADDING);
   error_code error = visit_number(iter, copy.get());
   return error;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("true");
-  if (!atomparsing::is_valid_true_atom(value)) { return T_ATOM_ERROR; }
+  // The non-length-aware validator reads a fixed 5 bytes; a malformed/truncated
+  // token at the very end of an unpadded buffer would over-read. Use the
+  // length-aware form there (the root variant already does this).
+  const bool ok = UNPADDED ? atomparsing::is_valid_true_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_true_atom(value);
+  if (!ok) { return T_ATOM_ERROR; }
   tape.append(0, internal::tape_type::TRUE_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("true");
   if (!atomparsing::is_valid_true_atom(value, iter.remaining_len())) { return T_ATOM_ERROR; }
   tape.append(0, internal::tape_type::TRUE_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("false");
-  if (!atomparsing::is_valid_false_atom(value)) { return F_ATOM_ERROR; }
+  const bool ok = UNPADDED ? atomparsing::is_valid_false_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_false_atom(value);
+  if (!ok) { return F_ATOM_ERROR; }
   tape.append(0, internal::tape_type::FALSE_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("false");
   if (!atomparsing::is_valid_false_atom(value, iter.remaining_len())) { return F_ATOM_ERROR; }
   tape.append(0, internal::tape_type::FALSE_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("null");
-  if (!atomparsing::is_valid_null_atom(value)) { return N_ATOM_ERROR; }
+  const bool ok = UNPADDED ? atomparsing::is_valid_null_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_null_atom(value);
+  if (!ok) { return N_ATOM_ERROR; }
   tape.append(0, internal::tape_type::NULL_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("null");
   if (!atomparsing::is_valid_null_atom(value, iter.remaining_len())) { return N_ATOM_ERROR; }
   tape.append(0, internal::tape_type::NULL_VALUE);
   return SUCCESS;
 }

+#if SIMDJSON_ENABLE_NAN_INF
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept {
+  iter.log_value("nan");
+  // For unpadded input use the length-aware validator so the 'infinity'-style
+  // 8-byte compare cannot read past the buffer on a malformed token at the end.
+  const bool ok = UNPADDED ? atomparsing::is_valid_nan_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_nan_atom(value);
+  if (!ok) { return errc; }
+  tape.append_double(std::numeric_limits<double>::quiet_NaN());
+  return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept {
+  iter.log_value("nan");
+  if (!atomparsing::is_valid_nan_atom(value, iter.remaining_len())) { return errc; }
+  tape.append_double(std::numeric_limits<double>::quiet_NaN());
+  return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_inf_atom(json_iterator &iter, const uint8_t *value) noexcept {
+  iter.log_value("inf");
+  // Because 'inf' is an extension, non a canonical atom, a tape error should be returned on failure.
+  // For unpadded input use the length-aware validator so the 'infinity' 8-byte
+  // compare cannot read past the buffer on a malformed token at the end.
+  const bool ok = UNPADDED ? atomparsing::is_valid_inf_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_inf_atom(value);
+  if (!ok) { return TAPE_ERROR; }
+  tape.append_double(std::numeric_limits<double>::infinity());
+  return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_inf_atom(json_iterator &iter, const uint8_t *value) noexcept {
+  iter.log_value("inf");
+  // Because 'inf' is an extension, non a canonical atom, a tape error should be returned on failure
+  if (!atomparsing::is_valid_inf_atom(value, iter.remaining_len())) { return TAPE_ERROR; }
+  tape.append_double(std::numeric_limits<double>::infinity());
+  return SUCCESS;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
 // private:

-simdjson_inline uint32_t tape_builder::next_tape_index(json_iterator &iter) const noexcept {
+template <bool UNPADDED>
+simdjson_inline uint32_t tape_builder_impl<UNPADDED>::next_tape_index(json_iterator &iter) const noexcept {
   return uint32_t(tape.next_tape_loc - iter.dom_parser.doc->tape.get());
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
   auto start_index = next_tape_index(iter);
   tape.append(start_index+2, start);
   tape.append(start_index, end);
   return SUCCESS;
 }

-simdjson_inline void tape_builder::start_container(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_inline void tape_builder_impl<UNPADDED>::start_container(json_iterator &iter) noexcept {
   iter.dom_parser.open_containers[iter.depth].tape_index = next_tape_index(iter);
   iter.dom_parser.open_containers[iter.depth].count = 0;
   tape.skip(); // We don't actually *write* the start element until the end.
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
   // Write the ending tape element, pointing at the start location
   const uint32_t start_tape_index = iter.dom_parser.open_containers[iter.depth].tape_index;
   tape.append(start_tape_index, end);
@@ -28165,13 +36828,15 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::end_container(json
   return SUCCESS;
 }

-simdjson_inline uint8_t *tape_builder::on_start_string(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_inline uint8_t *tape_builder_impl<UNPADDED>::on_start_string(json_iterator &iter) noexcept {
   // we advance the point, accounting for the fact that we have a NULL termination
   tape.append(current_string_buf_loc - iter.dom_parser.doc->string_buf.get(), internal::tape_type::STRING);
   return current_string_buf_loc + sizeof(uint32_t);
 }

-simdjson_inline void tape_builder::on_end_string(uint8_t *dst) noexcept {
+template <bool UNPADDED>
+simdjson_inline void tape_builder_impl<UNPADDED>::on_end_string(uint8_t *dst) noexcept {
   uint32_t str_length = uint32_t(dst - (current_string_buf_loc + sizeof(uint32_t)));
   // TODO check for overflow in case someone has a crazy string (>=4GB?)
   // But only add the overflow check when the document itself exceeds 4GB
@@ -28542,16 +37207,6 @@ simdjson_inline int count_ones(uint64_t input_num) {
 }
 #endif

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
-                                         uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
-  *result = value1 + value2;
-  return *result < value1;
-#else
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-#endif
-}

 } // unnamed namespace
 } // namespace ppc64
@@ -29450,6 +38105,9 @@ namespace atomparsing {
 // to the compile-time constant 1936482662.
 simdjson_inline uint32_t string_to_uint32(const char* str) { uint32_t val; std::memcpy(&val, str, sizeof(uint32_t)); return val; }

+// Acts on the same principle as string_to_uint32, but on an 8-byte block of memory
+simdjson_inline uint64_t string_to_uint64(const char* str) { uint64_t val; std::memcpy(&val, str, sizeof(uint64_t)); return val; }
+

 // Again in str4ncmp we use a memcpy to avoid undefined behavior. The memcpy may appear expensive.
 // Yet all decent optimizing compilers will compile memcpy to a single instruction, just about.
@@ -29461,6 +38119,28 @@ simdjson_inline uint32_t str4ncmp(const uint8_t *src, const char* atom) {
   return srcval ^ string_to_uint32(atom);
 }

+// Checks that the first 8 characters of the input string match the given atom in a case-insensitive manner.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint64_t str8ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+  uint64_t srcval; // we want to avoid unaligned 32-bit loads (undefined in C/C++)
+  static_assert(sizeof(uint64_t) <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be larger than 8 bytes");
+  std::memcpy(&srcval, src, sizeof(uint64_t));
+
+  return (srcval | 0x2020202020202020ull) ^ string_to_uint64(atom);
+}
+
+// Checks that the first 3 characters of 'src' match 'atom' in a case-insensitive way.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint32_t str3ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+  return ((src[0] | 0x20) ^ atom[0]) //
+       | ((src[1] | 0x20) ^ atom[1]) //
+       | ((src[2] | 0x20) ^ atom[2]);
+}
+
 simdjson_warn_unused
 simdjson_inline bool is_valid_true_atom(const uint8_t *src) {
   return (str4ncmp(src, "true") | jsoncharutils::is_not_structural_or_whitespace(src[4])) == 0;
@@ -29497,6 +38177,71 @@ simdjson_inline bool is_valid_null_atom(const uint8_t *src, size_t len) {
   else { return false; }
 }

+#if SIMDJSON_ENABLE_NAN_INF
+// "nan" is 3 bytes; we check characters and then verify the next
+// character is structural or whitespace. We accept both "nan" and "NaN".
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src) {
+  return (str3ncmp_case_insensitive(src, "nan")
+        | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+}
+
+// checks that the next four characters of a string are 'nan"', where the 'nan'
+// is checked in a case-insensitive way.
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_in_string(const uint8_t *src) {
+  return (str3ncmp_case_insensitive(src, "nan") | (src[3] ^ '"')) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src, size_t len) {
+  if (len > 3) { return is_valid_nan_atom(src); }
+  if (len == 3) { return str3ncmp_case_insensitive(src, "nan") == 0; }
+  return false;
+}
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src) {
+  bool is_short_inf = (str3ncmp_case_insensitive(src, "inf")
+                     | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+  if(is_short_inf) return true;
+
+  // Check for 'infinity' (any capitalization)
+  return (str8ncmp_case_insensitive(src, "infinity") | jsoncharutils::is_not_structural_or_whitespace(src[8])) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_in_string(const uint8_t *src) {
+  bool is_short_inf = (str3ncmp_case_insensitive(src, "inf") | (src[3] ^ '"')) == 0;
+  if(is_short_inf) return true;
+
+  return (str8ncmp_case_insensitive(src, "infinity") | (src[8] ^ '"')) == 0;
+}
+
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src, size_t len) {
+  if (len > 8) { return is_valid_inf_atom(src); }
+  if (len == 8 && str8ncmp_case_insensitive(src, "infinity") == 0) {
+    return true;
+  }
+  if (len > 3) {
+    return (str3ncmp_case_insensitive(src, "inf")
+          | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+  }
+  if (len == 3) { return str3ncmp_case_insensitive(src, "inf") == 0; }
+  return false;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
 } // namespace atomparsing
 } // unnamed namespace
 } // namespace ppc64
@@ -29587,7 +38332,7 @@ inline simdjson_warn_unused error_code dom_parser_implementation::set_capacity(s
 }

 inline simdjson_warn_unused error_code dom_parser_implementation::set_max_depth(size_t max_depth) noexcept {
-  if(max_depth == 0 || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
+  if(max_depth == 0 || max_depth > SIMDJSON_MAX_DEPTH || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
   // Stage 2 stacks
   open_containers.reset(new (std::nothrow) open_container[max_depth]);
   is_array.reset(new (std::nothrow) bool[max_depth]);
@@ -29766,6 +38511,7 @@ protected:
 /* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_NUMBERPARSING_H */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/base.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/jsoncharutils.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/atomparsing.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/internal/numberparsing_tables.h" */
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */

@@ -29805,6 +38551,18 @@ simdjson_inline double to_double(uint64_t mantissa, uint64_t real_exponent, bool
     return d;
 }

+// Convert a mantissa, an exponent and a sign bit into an ieee32 float (binary32).
+// The real_exponent needs to be in [0, 254] (technically real_exponent = 255 would be acceptable).
+// The mantissa should be in [0,1<<24). The bit at index (1U << 23) will be zeroed.
+simdjson_inline float to_float(uint32_t mantissa, uint32_t real_exponent, bool negative) {
+    float f;
+    mantissa &= ~(uint32_t(1) << 23);
+    mantissa |= real_exponent << 23;
+    mantissa |= ((static_cast<uint32_t>(negative)) << 31);
+    std::memcpy(&f, &mantissa, sizeof(f));
+    return f;
+}
+
 // Attempts to compute i * 10^(power) exactly; and if "negative" is
 // true, negate the result.
 // This function will only work in some cases, when it does not work, success is
@@ -30061,6 +38819,219 @@ simdjson_inline bool compute_float_64(int64_t power, uint64_t i, bool negative,
   return true;
 }

+// Attempts to compute i * 10^(power) as a binary32 (float) value; and if
+// "negative" is true, negate the result.
+//
+// This is the same Eisel-Lemire algorithm as compute_float_64, retargeted at
+// binary32. It is adapted from fast_float
+// (https://github.com/fastfloat/fast_float), where the routine is written once
+// and instantiated for each binary format. We only need 24 bits of mantissa
+// instead of 53, so we ask the truncated product for 24 + 2 = 26 accurate bits
+// (plus one bit that may be lost to the "upperbit" shift), and we shift the
+// product right by (upperbit + 64 - 23 - 3) instead of (upperbit + 64 - 52 - 3).
+//
+// The power-of-five table is shared with the binary64 code: the useful range
+// for binary32, [smallest_power_binary32, largest_power_binary32], is a strict
+// subset of [smallest_power, largest_power].
+//
+// The function returns false when the result would be infinite: simdjson
+// refuses to parse infinite values, so the caller reports an error. Unlike
+// compute_float_64, a false return never means "try harder": the accuracy of
+// the two-limb product is guaranteed by Noble Mushtak and Daniel Lemire,
+// "Fast Number Parsing Without Fallback" (https://arxiv.org/abs/2212.06644).
+// The caller must still fall back to a slow path when the 64-bit mantissa i
+// was truncated (more than 19 significant digits).
+//
+// We assume that power is in the [smallest_power, largest_power] interval: the
+// caller is responsible for this check.
+simdjson_inline bool compute_float_32(int64_t power, uint64_t i, bool negative, float &d) {
+  // Powers of ten that are exactly representable as binary32 values: 10^k is
+  // exact as long as k <= 10 (5^10 = 9765625 < 2**24 while 5^11 > 2**24).
+  static constexpr float power_of_ten_float[] = {
+      1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+      1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+  // The range of powers of ten that a non-zero, finite binary32 value can be
+  // built from. Anything smaller rounds to zero, anything larger is infinite.
+  // These are the smallest_power_of_ten()/largest_power_of_ten() constants that
+  // fast_float uses for binary32.
+  constexpr int smallest_power_binary32 = -65;
+  constexpr int largest_power_binary32 = 38;
+
+  // We start with the fast path described in
+  // Clinger WD. How to read floating point numbers accurately.
+  // ACM SIGPLAN Notices. 1990
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+  // We cannot be certain that x/y is rounded to nearest.
+  if (0 <= power && power <= 10 && i <= 16777215)
+#else
+  if (-10 <= power && power <= 10 && i <= 16777215)
+#endif
+  {
+    // Convert the integer into a float. This is lossless since
+    // 0 <= i <= 2^24 - 1.
+    d = float(i);
+    // Both d and the power of ten are exactly representable as binary32
+    // values, so the product (or quotient) is correctly rounded.
+    if (power < 0) {
+      d = d / power_of_ten_float[-power];
+    } else {
+      d = d * power_of_ten_float[power];
+    }
+    if (negative) {
+      d = -d;
+    }
+    return true;
+  }
+
+  // The fast path has failed, so we fall back on the Eisel-Lemire algorithm.
+  // It needs i > 0 (so that the leading bit of i can be normalized), so we
+  // handle i == 0 separately. We also handle the powers of ten that are so
+  // small (or so large) that the answer is zero (or infinite) whatever the
+  // mantissa is.
+  if (i == 0 || power < smallest_power_binary32) {
+    d = negative ? -0.0f : 0.0f;
+    return true;
+  }
+  if (power > largest_power_binary32) {
+    // We have, for sure, an infinite value.
+    return false;
+  }
+
+  // The exponent is 128 + 63 + power + floor(log(5**power)/log(2)).
+  // The 128 comes from the ieee32 standard (the minimal exponent is -127).
+  // The 63 comes from the fact that we use a 64-bit word.
+  // See compute_float_64 for a discussion of the magical 152170 + 65536.
+  int64_t exponent = (((152170 + 65536) * power) >> 16) + 128 + 63;
+
+  // We want the most significant bit of i to be 1. Shift if needed.
+  int lz = leading_zeroes(i);
+  i <<= lz;
+
+  // We want the most significant 64 bits of the product i * 5**power. It is
+  // safe to index the table because
+  // smallest_power <= smallest_power_binary32 <= power
+  //               <= largest_power_binary32 <= largest_power.
+  const uint32_t index = 2 * uint32_t(power - simdjson::internal::smallest_power);
+#if SIMDJSON_STATIC_REFLECTION
+  simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index]);
+#else
+  simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index]);
+#endif
+
+  // Unless the least significant 38 bits of the high (64-bit) part of the full
+  // product are all 1s, then we know that the most significant 26 bits are
+  // exact and no further work is needed. Having 26 bits is necessary because
+  // we need 24 bits for the mantissa but we have to have one rounding bit and
+  // we can waste a bit if the most significant bit of the product is zero.
+  // (For binary64, the same reasoning gives 55 bits and a 9-bit mask.)
+  if((firstproduct.high & 0x3FFFFFFFFF) == 0x3FFFFFFFFF) {
+    // The truncated multiplication was not accurate enough; use the next 64
+    // bits of the power of five to refine it. See compute_float_64 for a
+    // detailed discussion.
+#if SIMDJSON_STATIC_REFLECTION
+    simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index + 1]);
+#else
+    simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index + 1]);
+#endif
+    firstproduct.low += secondproduct.high;
+    if(secondproduct.high > firstproduct.low) { firstproduct.high++; }
+  }
+  uint64_t lower = firstproduct.low;
+  uint64_t upper = firstproduct.high;
+  // The final mantissa should be 24 bits with a leading 1.
+  // We shift it so that it occupies 25 bits with a leading 1.
+  ///////
+  uint64_t upperbit = upper >> 63;
+  uint64_t mantissa = upper >> (upperbit + 38); // 38 == 64 - 23 - 3
+  lz += int(1 ^ upperbit);
+
+  // Here we have mantissa < (1<<25).
+  int64_t real_exponent = exponent - lz;
+  if (simdjson_unlikely(real_exponent <= 0)) { // we have a subnormal?
+    // Here we have that real_exponent <= 0 so -real_exponent >= 0
+    if(-real_exponent + 1 >= 64) { // if we have more than 64 bits below the minimum exponent, you have a zero for sure.
+      d = negative ? -0.0f : 0.0f;
+      return true;
+    }
+    // next line is safe because -real_exponent + 1 < 64
+    mantissa >>= -real_exponent + 1;
+    // Thankfully, we can't have both "round-to-even" and subnormals because
+    // "round-to-even" only occurs for powers close to 0.
+    mantissa += (mantissa & 1); // round up
+    mantissa >>= 1;
+    // As in compute_float_64, rounding up may take us out of the subnormal
+    // range, so we can only decide after rounding.
+    real_exponent = (mantissa < (uint64_t(1) << 23)) ? 0 : 1;
+    d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+    return true;
+  }
+  // We have to round to even. The "to even" part is only a problem when we are
+  // right in between two floats, which we guard against. The bounds on the
+  // power of ten are those of fast_float's
+  // min_exponent_round_to_even()/max_exponent_round_to_even() for binary32:
+  // when q >= 0, (2m+1) must be divisible by 5^q with 5^q <= 2^25, so q <= 10;
+  // when q < 0, we need 2^24 x 5^{-q} < 2^{64}, so q >= -17.
+  if (simdjson_unlikely((lower <= 1) && (power >= -17) && (power <= 10) && ((mantissa & 3) == 1))) {
+    if((mantissa << (upperbit + 38)) == upper) {
+      mantissa &= ~uint64_t(1);   // flip it so that we do not round up
+    }
+  }
+
+  mantissa += mantissa & 1;
+  mantissa >>= 1;
+
+  // Here we have mantissa < (1<<24), unless there was an overflow
+  if (mantissa >= (uint64_t(1) << 24)) {
+    mantissa = (uint64_t(1) << 23);
+    real_exponent++;
+  }
+  mantissa &= ~(uint64_t(1) << 23);
+  // we have to check that real_exponent is in range, otherwise we bail out
+  if (simdjson_unlikely(real_exponent > 254)) {
+    // We have an infinite value!!! We could actually throw an error here if we could.
+    return false;
+  }
+  d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+  return true;
+}
+
+#if SIMDJSON_ENABLE_NAN_INF
+// Parses a nan or infinity. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, double& d) noexcept {
+  if (atomparsing::is_valid_inf_atom(src)) {
+    double inf = std::numeric_limits<double>::infinity();
+    d = negative ? -inf : inf;
+    return true;
+  }
+
+  if (atomparsing::is_valid_nan_atom(src)) {
+    d = std::numeric_limits<double>::quiet_NaN();
+    return true;
+  }
+
+  return false;
+}
+
+// Parses a nan or infinity as a binary32 value. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, float& d) noexcept {
+  if (atomparsing::is_valid_inf_atom(src)) {
+    float inf = std::numeric_limits<float>::infinity();
+    d = negative ? -inf : inf;
+    return true;
+  }
+
+  if (atomparsing::is_valid_nan_atom(src)) {
+    d = std::numeric_limits<float>::quiet_NaN();
+    return true;
+  }
+
+  return false;
+}
+#endif
+
 // We call a fallback floating-point parser that might be slow. Note
 // it will accept JSON numbers, but the JSON spec. is more restrictive so
 // before you call parse_float_fallback, you need to have validated the input
@@ -30098,6 +39069,16 @@ static bool parse_float_fallback(const uint8_t *ptr, const uint8_t *end_ptr, dou
   return !(*outDouble > (std::numeric_limits<double>::max)() || *outDouble < std::numeric_limits<double>::lowest());
 }

+// Same as parse_float_fallback, but for binary32 (float) values. Going through
+// the binary64 fallback and then rounding to binary32 would be subject to
+// double rounding, so we run the fallback algorithm directly on binary32.
+static bool parse_float_fallback(const uint8_t *ptr, float *outFloat) {
+  *outFloat = simdjson::internal::from_chars_float(reinterpret_cast<const char *>(ptr));
+  // We do not accept infinite values. See the binary64 version above for why we
+  // do not use std::isfinite.
+  return !(*outFloat > (std::numeric_limits<float>::max)() || *outFloat < std::numeric_limits<float>::lowest());
+}
+
 // check quickly whether the next 8 chars are made of digits
 // at a glance, it looks better than Mula's
 // http://0x80.pl/articles/swar-digits-validate.html
@@ -30116,6 +39097,28 @@ simdjson_inline bool is_made_of_eight_digits_fast(const uint8_t *chars) {
           0x3333333333333333);
 }

+// The same idea for four characters. A block of eight only bites when eight
+// digits are left, so a run of 4 to 7 digits used to fall back to one digit at a
+// time; taking four of them at once is what makes the tail of a long fraction
+// cheap. Adapted from fast_float, credit @aqrit.
+simdjson_inline bool is_made_of_four_digits_fast(const uint8_t *chars) {
+  uint32_t val;
+  // Reads up to 3 bytes beyond the digits, which SIMDJSON_PADDING covers.
+  static_assert(3 <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be bigger than 3");
+  std::memcpy(&val, chars, 4);
+  return !(((val + 0x46464646) | (val - 0x30303030)) & 0x80808080);
+}
+
+// Only call this when is_made_of_four_digits_fast() says the four characters are
+// digits. Little-endian: guarded by SIMDJSON_SWAR_NUMBER_PARSING.
+simdjson_inline uint32_t parse_four_digits_unrolled(const uint8_t *chars) {
+  uint32_t val;
+  std::memcpy(&val, chars, 4);
+  val -= 0x30303030;
+  val = (val * 10) + (val >> 8);
+  return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
 template<typename I>
 SIMDJSON_NO_SANITIZE_UNDEFINED // We deliberately allow overflow here and check later
 simdjson_inline bool parse_digit(const uint8_t c, I &i) {
@@ -30132,26 +39135,73 @@ simdjson_inline bool is_digit(const uint8_t c) {
   return static_cast<uint8_t>(c - '0') <= 9;
 }

-simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
-  // we continue with the fiction that we have an integer. If the
-  // floating point number is representable as x * 10^z for some integer
-  // z that fits in 53 bits, then we will be able to convert back the
-  // the integer into a float in a lossless manner.
-  const uint8_t *const first_after_period = p;
+// Consumes a run of digits into i, eight at a time while eight are available,
+// then four, then one at a time. Overflow is deliberate: the caller counts the
+// digits and falls back when there are too many for a 64-bit mantissa.
+//
+// Long runs of digits are what a float-heavy document is made of: the fraction
+// of a binary64 printed to full precision is 15 to 17 digits, and reading those
+// one at a time is the single largest cost in parse_double(). Integer parts are
+// left alone: they are usually a handful of digits, and there the check for a
+// block of eight is wasted work.
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_fraction_digits(const uint8_t *&p, uint64_t &i) {
+#ifdef SIMDJSON_SWAR_NUMBER_PARSING
+#if SIMDJSON_SWAR_NUMBER_PARSING
+  while (is_made_of_eight_digits_fast(p)) {
+    i = i * 100000000 + parse_eight_digits_unrolled(p);
+    p += 8;
+  }
+  // A 4 to 7 digit remainder is the common case once the blocks of eight are
+  // gone: 15 fraction digits leave 7, and 17 leave 1 after two blocks.
+  if (is_made_of_four_digits_fast(p)) {
+    i = i * 10000 + parse_four_digits_unrolled(p);
+    p += 4;
+  }
+#endif // SIMDJSON_SWAR_NUMBER_PARSING
+#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
+  while (parse_digit(*p, i)) { p++; }
+}

+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_float_integer_digits(const uint8_t *&p, uint64_t &i,
+                                                bool &leading_zero) {
+  if (!parse_digit(*p, i)) { leading_zero = true; return; }
+  p++;
+  leading_zero = (i == 0);
+  while (parse_digit(*p, i)) { p++; }
+}
+
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_integer_digits(const uint8_t *&p, uint64_t &i) {
 #ifdef SIMDJSON_SWAR_NUMBER_PARSING
 #if SIMDJSON_SWAR_NUMBER_PARSING
-  // this helps if we have lots of decimals!
-  // this turns out to be frequent enough.
+  // Identifiers, timestamps and counters often have eight digits or more.
+  // Prior related work: jsonifier parses integers as eight-digit SWAR words
+  // (str_to_i.hpp, https://github.com/nihilai-collective/Jsonifier). This
+  // takes one such word with parse_eight_digits_unrolled, the routine
+  // simdjson uses for long fractions.
   if (is_made_of_eight_digits_fast(p)) {
     i = i * 100000000 + parse_eight_digits_unrolled(p);
     p += 8;
   }
+  const uint8_t *const swar_end = p + 8;
+  while (p < swar_end && is_made_of_four_digits_fast(p)) {
+    i = i * 10000 + parse_four_digits_unrolled(p);
+    p += 4;
+  }
 #endif // SIMDJSON_SWAR_NUMBER_PARSING
 #endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
-  // Unrolling the first digit makes a small difference on some implementations (e.g. westmere)
-  if (parse_digit(*p, i)) { ++p; }
   while (parse_digit(*p, i)) { p++; }
+}
+
+simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
+  // we continue with the fiction that we have an integer. If the
+  // floating point number is representable as x * 10^z for some integer
+  // z that fits in 53 bits, then we will be able to convert back the
+  // the integer into a float in a lossless manner.
+  const uint8_t *const first_after_period = p;
+  parse_fraction_digits(p, i);
   exponent = first_after_period - p;
   // Decimal without digits (123.) is illegal
   if (exponent == 0) {
@@ -30240,7 +39290,7 @@ simdjson_inline size_t significant_digits(const uint8_t * start_digits, size_t d
 } // unnamed namespace

 /** @private */
-static error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
+inline error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
   if (parse_float_fallback(src, answer)) {
     return SUCCESS;
   }
@@ -30323,15 +39373,17 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
   return SUCCESS;              // always succeeds
 }

-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept  { return false; }
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept  { return false; }
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept { return number_type::signed_integer; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline bool is_negative(const uint8_t *) noexcept  { return false; }
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t *) noexcept  { return false; }
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t *) noexcept { return number_type::signed_integer; }
 #else

 // parse the number at src
@@ -30362,7 +39414,21 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
   size_t digit_count = size_t(p - start_digits);
-  if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) { return INVALID_NUMBER(src); }
+  if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // By this point, we know that our input does not begin with a digit. We will attempt
+    // to handle NaN/Infinity.
+
+    double d;
+    if (compute_nan_inf(p, negative, d)) {
+      writer.append_double(d);
+      return SUCCESS;
+    }
+#endif
+
+    return INVALID_NUMBER(src);
+  }

   //
   // Handle floats if there is a . or e (or both)
@@ -30411,7 +39477,7 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
     // - Therefore, if the number is positive and lower than that, it's overflow.
     // - The value we are looking at is less than or equal to INT64_MAX.
     //
-    }  else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return INVALID_NUMBER(src); }
+    }  else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return BIGINT_NUMBER(src); }
   }

   // Write unsigned if it does not fit in a signed integer.
@@ -30510,7 +39576,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const u
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -30608,7 +39674,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_stri
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -30663,7 +39729,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uin
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -30749,7 +39815,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = src;
   uint64_t i = 0;
-  while (parse_digit(*src, i)) { src++; }
+  parse_integer_digits(src, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -30789,11 +39855,20 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   //
   uint64_t i = 0;
   const uint8_t *p = src;
-  p += parse_digit(*p, i);
-  bool leading_zero = (i == 0);
-  while (parse_digit(*p, i)) { p++; }
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
   // no integer digits, or 0123 (zero must be solo)
-  if ( p == src ) { return INCORRECT_TYPE; }
+  if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no loading digits, the number may be nan or infinity.
+    // Attempt to compute those, and return on success.
+    double d;
+    if (compute_nan_inf(p, negative, d)) { return d; }
+#endif
+
+    return INCORRECT_TYPE;
+  }
   if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }

   //
@@ -30804,9 +39879,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   if (simdjson_likely(*p == '.')) {
     p++;
     const uint8_t *start_decimal_digits = p;
-    if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
-    p++;
-    while (parse_digit(*p, i)) { p++; }
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
     exponent = -(p - start_decimal_digits);

     // Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -30855,6 +39929,97 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   return d;
 }

+// Parse a JSON number into a binary32 (float) value.
+//
+// This mirrors parse_double, but it rounds to binary32 directly instead of
+// rounding to binary64 and then to binary32: the latter is subject to double
+// rounding and would not always produce the nearest float.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * src) noexcept {
+  //
+  // Check for minus sign
+  //
+  bool negative = (*src == '-');
+  src += uint8_t(negative);
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, the number may be nan or infinity.
+    // Attempt to compute those, and return on success.
+    float f;
+    if (compute_nan_inf(p, negative, f)) { return f; }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = p-start_digits > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if (*p == 'e' || *p == 'E') {
+    p++;
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while (parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  float d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_32(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
 simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
   return (*src == '-');
 }
@@ -31007,11 +40172,25 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   //
   uint64_t i = 0;
   const uint8_t *p = src;
-  p += parse_digit(*p, i);
-  bool leading_zero = (i == 0);
-  while (parse_digit(*p, i)) { p++; }
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
   // no integer digits, or 0123 (zero must be solo)
-  if ( p == src ) { return INCORRECT_TYPE; }
+  if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, attempt to parse numbers that are either
+    // NaN or Infinity
+    if (atomparsing::is_valid_inf_in_string(src)) {
+      double inf = std::numeric_limits<double>::infinity();
+      return negative ? -inf : inf;
+    }
+
+    if (atomparsing::is_valid_nan_in_string(src)) {
+      return std::numeric_limits<double>::quiet_NaN();
+    }
+#endif
+
+    return INCORRECT_TYPE;
+  }
   if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }

   //
@@ -31022,9 +40201,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   if (simdjson_likely(*p == '.')) {
     p++;
     const uint8_t *start_decimal_digits = p;
-    if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
-    p++;
-    while (parse_digit(*p, i)) { p++; }
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
     exponent = -(p - start_decimal_digits);

     // Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -31073,6 +40251,99 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   return d;
 }

+// Parse a JSON number held inside a JSON string into a binary32 (float) value.
+// See parse_float for why we do not simply round parse_double_in_string.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * src) noexcept {
+  //
+  // Check for minus sign
+  //
+  bool negative = (*(src + 1) == '-');
+  src += uint8_t(negative) + 1;
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, attempt to parse numbers that are either
+    // NaN or Infinity
+    if (atomparsing::is_valid_inf_in_string(src)) {
+      float inf = std::numeric_limits<float>::infinity();
+      return negative ? -inf : inf;
+    }
+
+    if (atomparsing::is_valid_nan_in_string(src)) {
+      return std::numeric_limits<float>::quiet_NaN();
+    }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = p-start_digits > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if (*p == 'e' || *p == 'E') {
+    p++;
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while (parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if (*p != '"') { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  float d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_32(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
 } // unnamed namespace
 #endif // SIMDJSON_SKIPNUMBERPARSING

@@ -31398,16 +40669,6 @@ simdjson_inline int count_ones(uint64_t input_num) {
 }
 #endif

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
-                                         uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
-  *result = value1 + value2;
-  return *result < value1;
-#else
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-#endif
-}

 } // unnamed namespace
 } // namespace ppc64
@@ -33103,6 +42364,296 @@ simdjson_inline uint32_t find_next_document_index(dom_parser_implementation &par
   return 0;
 }

+/**
+ * Sentinel value returned to indicate a document started but didn't fit
+ * (CAPACITY error), as opposed to 0 which means no document content found
+ * (EMPTY).
+ */
+constexpr uint32_t DOCUMENT_TOO_LARGE = UINT32_MAX;
+
+/**
+ * For RFC 7464 JSON text sequences, filter RS from structural indexes and
+ * find batch boundaries.
+ *
+ * In JSON sequence mode, RS (0x1E) marks the start of each JSON text.
+ * RS bytes appear in structural_indexes as they are classified as scalars.
+ * This function:
+ * 1. Scans structural_indexes to find and count RS positions
+ * 2. Filters RS out of structural_indexes in-place
+ * 3. Determines batch boundaries based on RS positions
+ *
+ * @param parser The parser with structural_indexes and buf.
+ * @param len The length of the current batch buffer.
+ * @param is_final True if this is the final batch (no more data coming).
+ * @param next_batch_start Output: offset where the next batch should start.
+ * @param scan_len Offset past which no value start may be derived. Equals len
+ *        unless stage 1 dropped a trailing unclosed string, whose bytes it
+ *        never validated.
+ * @return The number of structural indexes to keep (after RS filtering),
+ *         0 if no document content found (EMPTY),
+ *         or DOCUMENT_TOO_LARGE if a document started but didn't fit (CAPACITY).
+ */
+simdjson_inline uint32_t find_next_document_index_json_sequence(
+    dom_parser_implementation &parser,
+    size_t len,
+    bool is_final,
+    uint32_t &next_batch_start,
+    size_t scan_len) {
+  // Default: next batch starts at end of buffer
+  next_batch_start = uint32_t(len);
+
+  if (parser.n_structural_indexes == 0) { return 0; }
+
+  // Phase 1: Scan structural_indexes to find RS positions and handle them.
+  // RS marks the start of a JSON text. For objects/arrays, the '{' or '[' after RS
+  // is already in structural_indexes (it's an operator). For scalars like numbers,
+  // the digit following RS is NOT in structural_indexes because the scanner sees
+  // RS as a scalar, making the digit a scalar continuation, not a start.
+  // We must: (1) remove RS from structural_indexes, and (2) for scalars, add the
+  // actual value start position.
+  // The EOF sentinel: len, or where a discarded unclosed string starts.
+  const uint32_t sentinel = parser.structural_indexes[parser.n_structural_indexes];
+  uint32_t write_idx = 0;
+  uint32_t last_rs_pos = 0;
+  uint32_t rs_count = 0;
+
+  for (uint32_t read_idx = 0; read_idx < parser.n_structural_indexes; read_idx++) {
+    const uint32_t pos = parser.structural_indexes[read_idx];
+    if (parser.buf[pos] == 0x1E) {
+      // This is an RS character - find the actual JSON value start.
+      last_rs_pos = pos;
+      rs_count++;
+      // Skip past this RS and any whitespace *and any additional RSes*
+      // to locate the real value. Consecutive RSes are degenerate
+      // "empty records" per RFC 7464; we collapse them here. They do
+      // not always appear as separate entries in structural_indexes
+      // because the scanner groups runs of adjacent non-whitespace
+      // scalar bytes (including RS) into a single scalar start.
+      uint32_t value_start = pos + 1;
+      while (value_start < scan_len) {
+        const uint8_t c = parser.buf[value_start];
+        if (c == ' ' || c == '\t' || c == '\n' || c == '\r') {
+          value_start++;
+        } else if (c == 0x1E) {
+          // Collapsed empty record. Still count it so rs_count reflects
+          // the true number of record markers and last_rs_pos tracks
+          // the final one.
+          last_rs_pos = value_start;
+          rs_count++;
+          value_start++;
+        } else {
+          break;
+        }
+      }
+      // If the scanner emitted additional structurals inside the
+      // whitespace+RS run we just walked over (i.e., isolated RSes
+      // separated by whitespace), skip past them so we do not
+      // double-count or double-emit.
+      while (read_idx + 1 < parser.n_structural_indexes &&
+             parser.structural_indexes[read_idx + 1] < value_start) {
+        read_idx++;
+      }
+      // Check if the value start is an operator (always present in
+      // scanner structural_indexes) or a scalar-like start (which may
+      // be missing from structural_indexes and must be added here).
+      // Note: '"' is NOT always in structural_indexes. The scanner
+      // classifies '"' as a scalar character and emits it as a
+      // structural only when it is a *scalar start* (preceded by
+      // whitespace or an operator). When '"' immediately follows an
+      // RS (which the scanner also classifies as scalar), it is
+      // treated as a scalar continuation and not emitted - so we
+      // must add it here just like any other scalar value.
+      if (value_start < scan_len) {
+        const uint8_t c = parser.buf[value_start];
+        const bool is_operator =
+            (c == '{' || c == '}' || c == '[' || c == ']' ||
+             c == ':' || c == ',');
+        // If the next scanner structural is exactly at value_start,
+        // the scanner already emitted it (it followed whitespace) and
+        // we must not add a duplicate - a subsequent iteration will
+        // copy it into write_idx.
+        const bool already_emitted =
+            (read_idx + 1 < parser.n_structural_indexes &&
+             parser.structural_indexes[read_idx + 1] == value_start);
+        if (!is_operator && !already_emitted) {
+          // Scalar value (number/true/false/null/string) - add its
+          // position since scanner missed it.
+          parser.structural_indexes[write_idx++] = value_start;
+        }
+      }
+    } else {
+      // Not RS, copy to output
+      parser.structural_indexes[write_idx++] = pos;
+    }
+  }
+
+  // Update structural index count. Compaction left a stale index in the slot
+  // past the end: restore the EOF sentinel that stage 1 had planted there, which
+  // document_stream::truncated_bytes() reads after a final batch.
+  parser.n_structural_indexes = write_idx;
+  parser.structural_indexes[write_idx] = sentinel;
+
+  if (parser.n_structural_indexes == 0) {
+    // Only RS markers here: the last one opens a record continuing past the
+    // window, so that is where the next batch resumes.
+    if (!is_final && rs_count > 0) { next_batch_start = last_rs_pos; }
+    return 0;
+  }
+  if (rs_count == 0) {
+    // No RS found; for final batch, try generic boundary detection
+    return is_final ? find_next_document_index(parser) : 0;
+  }
+
+  // Phase 2: Determine batch boundaries based on RS positions
+
+  if (is_final) {
+    // Final batch: all documents are complete (last one ends at EOF).
+    // In json_sequence mode, RS markers define document boundaries, so all
+    // remaining structurals form complete documents. Return them all directly.
+    // (Calling find_next_document_index() would fail for scalar documents.)
+    return parser.n_structural_indexes;
+  }
+
+  // Partial batch: need to find complete documents only.
+  // A document starting at an RS is complete if there is another RS after it.
+  next_batch_start = last_rs_pos;
+
+  if (rs_count < 2) {
+    // Only one RS, so we have at most one document that may be incomplete.
+    // We cannot confirm it is complete without another RS.
+    // Return DOCUMENT_TOO_LARGE if content was found (write_idx > 0), 0 if only separators.
+    return (parser.n_structural_indexes > 0) ? DOCUMENT_TOO_LARGE : 0;
+  }
+
+  // We have at least 2 RS markers. The last complete document ends before last_rs_pos.
+
+  // Find the structural index cutoff: keep only structurals < last_rs_pos.
+  // Since we already filtered RS, all remaining structurals are valid.
+  // We iterate backward to find the last structural before last_rs_pos.
+  uint32_t keep_count = 0;
+  for (uint32_t i = parser.n_structural_indexes; i > 0; i--) {
+    if (parser.structural_indexes[i - 1] < last_rs_pos) {
+      keep_count = i;
+      break;
+    }
+  }
+
+  // No structurals before the last RS - no complete documents
+  if (keep_count == 0) { return 0; }
+
+  // All documents before the last RS are complete by definition (the next RS
+  // confirms their end). No need to call find_next_document_index() which
+  // would fail for scalar documents like `1` or `"hello"`.
+  return keep_count;
+}
+
+/**
+ * Filter comma-delimited documents by removing root-level commas from
+ * structural indexes.
+ *
+ * For comma-delimited format like `{...},{...},{...}`, we need to remove
+ * the commas that separate documents (depth 0) while preserving commas
+ * inside arrays and objects (depth > 0).
+ *
+ * After filtering, the structural indexes look like whitespace-delimited
+ * documents, so find_next_document_index() works unchanged.
+ *
+ * @param parser The parser with structural_indexes and buf.
+ * @param len The length of the current batch buffer.
+ * @param is_final True if this is the final batch (no more data coming).
+ * @param next_batch_start Output: offset where the next batch should start.
+ * @return The number of structural indexes to keep,
+ *         0 if no document content found (EMPTY),
+ *         or DOCUMENT_TOO_LARGE if a document started but didn't fit (CAPACITY).
+ */
+simdjson_inline uint32_t filter_comma_delimited(
+    dom_parser_implementation &parser,
+    size_t len,
+    bool is_final,
+    uint32_t &next_batch_start) {
+  // Default: next batch starts at end of buffer
+  next_batch_start = uint32_t(len);
+
+  if (parser.n_structural_indexes == 0) { return 0; }
+
+  // Track depth to identify root-level commas (depth 0)
+  int depth = 0;
+  // The EOF sentinel: len, or where a discarded unclosed string starts.
+  const uint32_t sentinel = parser.structural_indexes[parser.n_structural_indexes];
+  uint32_t write_idx = 0;
+  uint32_t last_root_comma_pos = 0;
+  uint32_t root_comma_count = 0;
+
+  for (uint32_t i = 0; i < parser.n_structural_indexes; i++) {
+    uint32_t idx = parser.structural_indexes[i];
+    uint8_t c = parser.buf[idx];
+
+    switch (c) {
+      case '{': case '[':
+        depth++;
+        parser.structural_indexes[write_idx++] = idx;
+        break;
+      case '}': case ']':
+        depth--;
+        parser.structural_indexes[write_idx++] = idx;
+        break;
+      case ',':
+        if (depth == 0) {
+          // Root-level comma = document boundary, skip it
+          last_root_comma_pos = idx;
+          root_comma_count++;
+          continue;
+        }
+        parser.structural_indexes[write_idx++] = idx;
+        break;
+      default:
+        // Colons, scalars, etc.
+        parser.structural_indexes[write_idx++] = idx;
+        break;
+    }
+  }
+
+  // Update structural index count. Compaction left a stale index in the slot
+  // past the end: restore the EOF sentinel that stage 1 had planted there, which
+  // document_stream::truncated_bytes() reads after a final batch.
+  parser.n_structural_indexes = write_idx;
+  parser.structural_indexes[write_idx] = sentinel;
+
+  if (parser.n_structural_indexes == 0) { return 0; }
+
+  if (is_final) {
+    // Final batch: use standard boundary detection on filtered indexes
+    return find_next_document_index(parser);
+  }
+
+  // Partial batch: need to find complete documents only.
+  // A document ending with a root comma is complete.
+  if (root_comma_count == 0) {
+    // No root commas found; we cannot confirm any document is complete.
+    // The whole batch might be one incomplete document.
+    // Return DOCUMENT_TOO_LARGE if content was found (write_idx > 0), 0 if only commas.
+    return (parser.n_structural_indexes > 0) ? DOCUMENT_TOO_LARGE : 0;
+  }
+
+  // We have at least one root comma. Documents before the last comma are complete.
+  next_batch_start = last_root_comma_pos + 1;
+
+  // Find the structural index cutoff: keep only structurals < last_root_comma_pos
+  uint32_t keep_count = 0;
+  for (uint32_t i = parser.n_structural_indexes; i > 0; i--) {
+    if (parser.structural_indexes[i - 1] < last_root_comma_pos) {
+      keep_count = i;
+      break;
+    }
+  }
+
+  if (keep_count == 0) { return 0; }
+
+  // Use standard boundary detection on the complete portion
+  parser.n_structural_indexes = keep_count;
+  return find_next_document_index(parser);
+}
+
 } // namespace stage1
 } // unnamed namespace
 } // namespace ppc64
@@ -33536,7 +43087,6 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
         return EMPTY;
       }
     }
-
     parser.n_structural_indexes = new_structural_indexes;
   } else if (partial == stage1_mode::streaming_final) {
     if(have_unclosed_string) { parser.n_structural_indexes--; }
@@ -33564,6 +43114,75 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
         // the trailing garbage.
         return EMPTY;
     }
+  } else if (partial == stage1_mode::json_sequence_partial) {
+    // RFC 7464: use RS positions for batch boundaries
+    // A discarded unclosed string also caps how far the filter may scan.
+    size_t scan_len = len;
+    if(have_unclosed_string) {
+      parser.n_structural_indexes--;
+      if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return CAPACITY; }
+      scan_len = parser.structural_indexes[parser.n_structural_indexes];
+    }
+    uint32_t next_batch_start = uint32_t(len);
+    auto new_structural_indexes = find_next_document_index_json_sequence(parser, len, false, next_batch_start, scan_len);
+    if (new_structural_indexes == DOCUMENT_TOO_LARGE) {
+      return CAPACITY;
+    }
+    if (new_structural_indexes == 0) {
+      // An EMPTY batch must still advance next_batch_start, or document_stream
+      // re-parses the same bytes forever. CAPACITY when it cannot.
+      if (next_batch_start == 0) { return CAPACITY; }
+      parser.n_structural_indexes = 0;
+      parser.structural_indexes[0] = next_batch_start;
+      return EMPTY;
+    }
+    parser.n_structural_indexes = new_structural_indexes;
+    parser.structural_indexes[parser.n_structural_indexes] = next_batch_start;
+  } else if (partial == stage1_mode::json_sequence_final) {
+    // RFC 7464: final batch, last document extends to EOF
+    // As above: a discarded unclosed string caps how far the filter may scan.
+    size_t scan_len = len;
+    if(have_unclosed_string) {
+      parser.n_structural_indexes--;
+      scan_len = parser.structural_indexes[parser.n_structural_indexes];
+    }
+    uint32_t next_batch_start = uint32_t(len);
+    parser.n_structural_indexes = find_next_document_index_json_sequence(parser, len, true, next_batch_start, scan_len);
+    // The filter compacted structural_indexes in place and restored the EOF
+    // sentinel past the compacted end, so the copy below is either the start
+    // of a truncated document or len, as in streaming_final.
+    parser.structural_indexes[parser.n_structural_indexes + 1] = parser.structural_indexes[parser.n_structural_indexes];
+    parser.structural_indexes[parser.n_structural_indexes] = uint32_t(len);
+    if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return EMPTY; }
+  } else if (partial == stage1_mode::comma_delimited_partial) {
+    // Comma-delimited: filter root-level commas, use comma positions for batch boundaries
+    if(have_unclosed_string) {
+      parser.n_structural_indexes--;
+      if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return CAPACITY; }
+    }
+    uint32_t next_batch_start = uint32_t(len);
+    auto new_structural_indexes = filter_comma_delimited(parser, len, false, next_batch_start);
+    if (new_structural_indexes == DOCUMENT_TOO_LARGE) {
+      return CAPACITY;
+    }
+    if (new_structural_indexes == 0) {
+      // An EMPTY batch must still advance next_batch_start, or document_stream
+      // re-parses the same bytes forever. CAPACITY when it cannot.
+      if (next_batch_start == 0) { return CAPACITY; }
+      parser.n_structural_indexes = 0;
+      parser.structural_indexes[0] = next_batch_start;
+      return EMPTY;
+    }
+    parser.n_structural_indexes = new_structural_indexes;
+    parser.structural_indexes[parser.n_structural_indexes] = next_batch_start;
+  } else if (partial == stage1_mode::comma_delimited_final) {
+    // Comma-delimited: final batch, last document extends to EOF
+    if(have_unclosed_string) { parser.n_structural_indexes--; }
+    uint32_t next_batch_start = uint32_t(len);
+    parser.n_structural_indexes = filter_comma_delimited(parser, len, true, next_batch_start);
+    parser.structural_indexes[parser.n_structural_indexes + 1] = parser.structural_indexes[parser.n_structural_indexes];
+    parser.structural_indexes[parser.n_structural_indexes] = uint32_t(len);
+    if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return EMPTY; }
   }
   checker.check_eof();
   return checker.errors();
@@ -33646,7 +43265,6 @@ namespace {
 namespace stage2 {

 class json_iterator;
-class structural_iterator;
 struct tape_builder;
 struct tape_writer;

@@ -33943,7 +43561,7 @@ public:
    *
    * - increment_count(iter) - each time a value is found in an array or object.
    */
-  template<bool STREAMING, typename V>
+  template<bool STREAMING, bool UNPADDED, typename V>
   simdjson_warn_unused simdjson_inline error_code walk_document(V &visitor) noexcept;

   /**
@@ -33960,6 +43578,7 @@ public:
    *
    * They may include invalid JSON as well (such as `1.2.3` or `ture`).
    */
+  template<bool UNPADDED>
   simdjson_inline const uint8_t *peek() const noexcept;
   /**
    * Advance to the next token.
@@ -33968,6 +43587,7 @@ public:
    *
    * They may include invalid JSON as well (such as `1.2.3` or `ture`).
    */
+  template<bool UNPADDED>
   simdjson_inline const uint8_t *advance() noexcept;
   /**
    * Get the remaining length of the document, from the start of the current token.
@@ -34016,7 +43636,7 @@ public:
   simdjson_warn_unused simdjson_inline error_code visit_primitive(V &visitor, const uint8_t *value) noexcept;
 };

-template<bool STREAMING, typename V>
+template<bool STREAMING, bool UNPADDED, typename V>
 simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &visitor) noexcept {
   logger::log_start();

@@ -34031,7 +43651,7 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &
   // Read first value
   //
   {
-    auto value = advance();
+    auto value = advance<UNPADDED>();

     // Make sure the outer object or array is closed before continuing; otherwise, there are ways we
     // could get into memory corruption. See https://github.com/simdjson/simdjson/issues/906
@@ -34043,8 +43663,8 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &
     }

     switch (*value) {
-      case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
-      case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+      case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+      case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
       default: SIMDJSON_TRY( visitor.visit_root_primitive(*this, value) ); break;
     }
   }
@@ -34061,29 +43681,29 @@ object_begin:
   SIMDJSON_TRY( visitor.visit_object_start(*this) );

   {
-    auto key = advance();
+    auto key = advance<UNPADDED>();
     if (*key != '"') { log_error("Object does not start with a key"); return TAPE_ERROR; }
     SIMDJSON_TRY( visitor.increment_count(*this) );
     SIMDJSON_TRY( visitor.visit_key(*this, key) );
   }

 object_field:
-  if (simdjson_unlikely( *advance() != ':' )) { log_error("Missing colon after key in object"); return TAPE_ERROR; }
+  if (simdjson_unlikely( *advance<UNPADDED>() != ':' )) { log_error("Missing colon after key in object"); return TAPE_ERROR; }
   {
-    auto value = advance();
+    auto value = advance<UNPADDED>();
     switch (*value) {
-      case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
-      case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+      case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+      case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
       default: SIMDJSON_TRY( visitor.visit_primitive(*this, value) ); break;
     }
   }

 object_continue:
-  switch (*advance()) {
+  switch (*advance<UNPADDED>()) {
     case ',':
       SIMDJSON_TRY( visitor.increment_count(*this) );
       {
-        auto key = advance();
+        auto key = advance<UNPADDED>();
         if (simdjson_unlikely( *key != '"' )) { log_error("Key string missing at beginning of field in object"); return TAPE_ERROR; }
         SIMDJSON_TRY( visitor.visit_key(*this, key) );
       }
@@ -34111,16 +43731,16 @@ array_begin:

 array_value:
   {
-    auto value = advance();
+    auto value = advance<UNPADDED>();
     switch (*value) {
-      case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
-      case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+      case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+      case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
       default: SIMDJSON_TRY( visitor.visit_primitive(*this, value) ); break;
     }
   }

 array_continue:
-  switch (*advance()) {
+  switch (*advance<UNPADDED>()) {
     case ',': SIMDJSON_TRY( visitor.increment_count(*this) ); goto array_value;
     case ']': log_end_value("array"); SIMDJSON_TRY( visitor.visit_array_end(*this) ); goto scope_end;
     default: log_error("Missing comma between array values"); return TAPE_ERROR;
@@ -34148,11 +43768,29 @@ simdjson_inline json_iterator::json_iterator(dom_parser_implementation &_dom_par
     dom_parser{_dom_parser} {
 }

+// Stage 1 leaves a sentinel index of value `len`; reading buf[len] requires
+// padding. For UNPADDED, return a dummy so walk_document can fail cleanly (#2815).
+template<bool UNPADDED>
 simdjson_inline const uint8_t *json_iterator::peek() const noexcept {
-  return &buf[*(next_structural)];
+  const uint32_t idx = *(next_structural);
+  SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+    if (simdjson_unlikely(idx >= dom_parser.len)) {
+      static constexpr uint8_t unpadded_eof_sentinel = 0;
+      return &unpadded_eof_sentinel;
+    }
+  }
+  return &buf[idx];
 }
+template<bool UNPADDED>
 simdjson_inline const uint8_t *json_iterator::advance() noexcept {
-  return &buf[*(next_structural++)];
+  const uint32_t idx = *(next_structural++);
+  SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+    if (simdjson_unlikely(idx >= dom_parser.len)) {
+      static constexpr uint8_t unpadded_eof_sentinel = 0;
+      return &unpadded_eof_sentinel;
+    }
+  }
+  return &buf[idx];
 }
 simdjson_inline size_t json_iterator::remaining_len() const noexcept {
   return dom_parser.len - *(next_structural-1);
@@ -34192,7 +43830,20 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::visit_root_primit
     case '"': return visitor.visit_root_string(*this, value);
     case 't': return visitor.visit_root_true_atom(*this, value);
     case 'f': return visitor.visit_root_false_atom(*this, value);
+#if SIMDJSON_ENABLE_NAN_INF
+    case 'n': {
+      auto err = visitor.visit_root_null_atom(*this, value);
+      if (err == SUCCESS) { return err; }
+      // propagate the error value returned by a bad 'null' atom if parsing 'nan' fails
+      return visitor.visit_root_nan_atom(*this, value, err);
+    }
+    // 'N' isn't a canonically recognized atom, so we return a TAPE_ERROR if failure occurs
+    case 'N': return visitor.visit_root_nan_atom(*this, value, TAPE_ERROR);
+    case 'i':
+    case 'I': return visitor.visit_root_inf_atom(*this, value);
+#else
     case 'n': return visitor.visit_root_null_atom(*this, value);
+#endif
     case '-':
     case '0': case '1': case '2': case '3': case '4':
     case '5': case '6': case '7': case '8': case '9':
@@ -34214,7 +43865,20 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::visit_primitive(V
   switch (*value) {
     case 't': return visitor.visit_true_atom(*this, value);
     case 'f': return visitor.visit_false_atom(*this, value);
+#if SIMDJSON_ENABLE_NAN_INF
+    case 'n': {
+      auto err = visitor.visit_null_atom(*this, value);
+      if (err == SUCCESS) { return err; }
+      // propagate the error value returned by a bad 'null' atom if parsing 'nan' fails
+      return visitor.visit_nan_atom(*this, value, err);
+    }
+    // 'N' isn't a canonically recognized atom, so we return a TAPE_ERROR if failure occurs
+    case 'N': return visitor.visit_nan_atom(*this, value, TAPE_ERROR);
+    case 'i':
+    case 'I': return visitor.visit_inf_atom(*this, value);
+#else
     case 'n': return visitor.visit_null_atom(*this, value);
+#endif
     default:
       log_error("Non-value found when value was expected!");
       return TAPE_ERROR;
@@ -34400,9 +44064,13 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_string(const uint8_t *src, u
            within the unicode codepoint handling code. */
         src += bs_dist;
         dst += bs_dist;
-        if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
-          return nullptr;
-        }
+        // Decode adjacent Unicode escapes without returning to the
+        // quote-and-backslash scanner between code points.
+        do {
+          if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
+            return nullptr;
+          }
+        } while (src[0] == '\\' && src[1] == 'u');
       } else {
         /* simple 1:1 conversion. Will eat bs_dist+2 characters in input and
          * write bs_dist+1 characters to output
@@ -34425,6 +44093,77 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_string(const uint8_t *src, u
   }
 }

+/**
+ * Bounds-safe variant of parse_string for input buffers that are NOT padded to
+ * len + SIMDJSON_PADDING bytes. `buf_end` is one past the last readable input
+ * byte (buf + len). It behaves exactly like parse_string while we are at least
+ * SIMDJSON_PADDING bytes away from buf_end (so every speculative SIMD read stays
+ * in bounds); once within the final SIMDJSON_PADDING bytes it copies the few
+ * remaining bytes into a space-padded scratch buffer and finishes with the
+ * regular parse_string. This keeps the delicate escape/Unicode handling in one
+ * place (the proven parse_string) rather than duplicating it.
+ *
+ * Correctness relies on stage 1 having validated the string, i.e. there is an
+ * unescaped closing quote within [src, buf_end); that quote is therefore inside
+ * the copied scratch, so parse_string finds it without running off the scratch.
+ */
+simdjson_warn_unused simdjson_inline uint8_t *parse_string_safe(const uint8_t *src, uint8_t *dst, bool allow_replacement, const uint8_t *buf_end) {
+  // Far from the end: identical to parse_string's loop. The guard uses
+  // SIMDJSON_PADDING (>= BYTES_PROCESSED) so copy_and_find never reads past
+  // buf_end; escape/Unicode look-aheads read within the string (before the
+  // closing quote, which is < buf_end), so they are in bounds here too.
+  // We add margin (+12) for handle_unicode_codepoint's worst-case lookahead:
+  // after seeing a high surrogate, it does hex_to_u32_nocheck on the immediate
+  // following bytes (+6 from the '\'), then (if it sees \u) another
+  // hex_to_u32_nocheck at +8..+11 relative to the backslash that started the
+  // escape. With bs_dist up to BYTES_PROCESSED-1 this reaches +11 from the
+  // chunk start. The +12 margin ensures that even on kernels where
+  // BYTES_PROCESSED == SIMDJSON_PADDING (e.g. icelake) the 4-byte read stays
+  // in-bounds. The scratch fallback (3*PAD) is already safe.
+  while (src + SIMDJSON_PADDING + 12 <= buf_end) {
+    auto b = backslash_and_quote{};
+    auto bs_quote = b.copy_and_find(src, dst);
+    if (bs_quote.has_quote_first()) {
+      return dst + bs_quote.quote_index();
+    }
+    if (bs_quote.has_backslash()) {
+      auto bs_dist = bs_quote.backslash_index();
+      uint8_t escape_char = src[bs_dist + 1];
+      if (escape_char == 'u') {
+        src += bs_dist;
+        dst += bs_dist;
+        if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
+          return nullptr;
+        }
+      } else {
+        uint8_t escape_result = escape_map[escape_char];
+        if (escape_result == 0u) {
+          return nullptr;
+        }
+        dst[bs_dist] = escape_result;
+        src += bs_dist + 2;
+        dst += bs_dist + 1;
+      }
+    } else {
+      src += backslash_and_quote::BYTES_PROCESSED;
+      dst += backslash_and_quote::BYTES_PROCESSED;
+    }
+  }
+  // Within the final SIMDJSON_PADDING bytes: copy what remains into a
+  // space-padded scratch (spaces are neither quote nor backslash, so they do not
+  // disturb matching) and let the regular parser finish from there. The closing
+  // quote is within `remaining` (< SIMDJSON_PADDING), so parse_string finds it in
+  // the chunk starting at some offset <= remaining and reads at most
+  // BYTES_PROCESSED (<= SIMDJSON_PADDING) further -- i.e. under 2*SIMDJSON_PADDING.
+  // We size at 3x for a comfortable margin (the unicode look-ahead reads a few
+  // extra bytes past an escape).
+  uint8_t scratch[SIMDJSON_PADDING * 3];
+  const size_t remaining = size_t(buf_end - src); // < SIMDJSON_PADDING
+  std::memset(scratch, ' ', sizeof(scratch));
+  std::memcpy(scratch, src, remaining);
+  return parse_string(scratch, dst, allow_replacement);
+}
+
 simdjson_warn_unused simdjson_inline uint8_t *parse_wobbly_string(const uint8_t *src, uint8_t *dst) {
   // It is not ideal that this function is nearly identical to parse_string.
   while (1) {
@@ -34479,73 +44218,6 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_wobbly_string(const uint8_t

 #endif // SIMDJSON_SRC_GENERIC_STAGE2_STRINGPARSING_H
 /* end file generic/stage2/stringparsing.h for ppc64 */
-/* including generic/stage2/structural_iterator.h for ppc64: #include <generic/stage2/structural_iterator.h> */
-/* begin file generic/stage2/structural_iterator.h for ppc64 */
-#ifndef SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H
-
-/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
-/* amalgamation skipped (editor-only): #define SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H */
-/* amalgamation skipped (editor-only): #include <generic/stage2/base.h> */
-/* amalgamation skipped (editor-only): #include <simdjson/generic/dom_parser_implementation.h> */
-/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
-
-namespace simdjson {
-namespace ppc64 {
-namespace {
-namespace stage2 {
-
-class structural_iterator {
-public:
-  const uint8_t* const buf;
-  uint32_t *next_structural;
-  dom_parser_implementation &dom_parser;
-
-  // Start a structural
-  simdjson_inline structural_iterator(dom_parser_implementation &_dom_parser, size_t start_structural_index)
-    : buf{_dom_parser.buf},
-      next_structural{&_dom_parser.structural_indexes[start_structural_index]},
-      dom_parser{_dom_parser} {
-  }
-  // Get the buffer position of the current structural character
-  simdjson_inline const uint8_t* current() {
-    return &buf[*(next_structural-1)];
-  }
-  // Get the current structural character
-  simdjson_inline char current_char() {
-    return buf[*(next_structural-1)];
-  }
-  // Get the next structural character without advancing
-  simdjson_inline char peek_next_char() {
-    return buf[*next_structural];
-  }
-  simdjson_inline const uint8_t* peek() {
-    return &buf[*next_structural];
-  }
-  simdjson_inline const uint8_t* advance() {
-    return &buf[*(next_structural++)];
-  }
-  simdjson_inline char advance_char() {
-    return buf[*(next_structural++)];
-  }
-  simdjson_inline size_t remaining_len() {
-    return dom_parser.len - *(next_structural-1);
-  }
-
-  simdjson_inline bool at_end() {
-    return next_structural == &dom_parser.structural_indexes[dom_parser.n_structural_indexes];
-  }
-  simdjson_inline bool at_beginning() {
-    return next_structural == dom_parser.structural_indexes.get();
-  }
-};
-
-} // namespace stage2
-} // unnamed namespace
-} // namespace ppc64
-} // namespace simdjson
-
-#endif // SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H
-/* end file generic/stage2/structural_iterator.h for ppc64 */
 /* including generic/stage2/tape_builder.h for ppc64: #include <generic/stage2/tape_builder.h> */
 /* begin file generic/stage2/tape_builder.h for ppc64 */
 #ifndef SIMDJSON_SRC_GENERIC_STAGE2_TAPE_BUILDER_H
@@ -34568,12 +44240,8 @@ namespace ppc64 {
 namespace {
 namespace stage2 {

-struct tape_builder {
-  template<bool STREAMING>
-  simdjson_warn_unused static simdjson_inline error_code parse_document(
-    dom_parser_implementation &dom_parser,
-    dom::document &doc) noexcept;
-
+template <bool UNPADDED>
+struct tape_builder_impl {
   /** Called when a non-empty document starts. */
   simdjson_warn_unused simdjson_inline error_code visit_document_start(json_iterator &iter) noexcept;
   /** Called when a non-empty document ends without error. */
@@ -34626,88 +44294,130 @@ struct tape_builder {
   simdjson_warn_unused simdjson_inline error_code visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept;
   simdjson_warn_unused simdjson_inline error_code visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept;

+#if SIMDJSON_ENABLE_NAN_INF
+  simdjson_warn_unused simdjson_inline error_code visit_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept;
+  simdjson_warn_unused simdjson_inline error_code visit_root_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept;
+  // Attempts to parse 'inf' or 'infinity' (case insensitive). Because neither are canonical atoms,
+  // this returns a tape error on failure.
+  simdjson_warn_unused simdjson_inline error_code visit_inf_atom(json_iterator &iter, const uint8_t *value) noexcept;
+  simdjson_warn_unused simdjson_inline error_code visit_root_inf_atom(json_iterator &iter, const uint8_t *value) noexcept;
+#endif
+
   /** Called each time a new field or element in an array or object is found. */
   simdjson_warn_unused simdjson_inline error_code increment_count(json_iterator &iter) noexcept;

   /** Next location to write to tape */
   tape_writer tape;
+public:
+  simdjson_inline tape_builder_impl(dom::document &doc) noexcept;
 private:
   /** Next write location in the string buf for stage 2 parsing */
   uint8_t *current_string_buf_loc;

-  simdjson_inline tape_builder(dom::document &doc) noexcept;
-
   simdjson_inline uint32_t next_tape_index(json_iterator &iter) const noexcept;
   simdjson_inline void start_container(json_iterator &iter) noexcept;
   simdjson_warn_unused simdjson_inline error_code end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept;
   simdjson_warn_unused simdjson_inline error_code empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept;
   simdjson_inline uint8_t *on_start_string(json_iterator &iter) noexcept;
   simdjson_inline void on_end_string(uint8_t *dst) noexcept;
-}; // struct tape_builder
+}; // struct tape_builder_impl

-template<bool STREAMING>
-simdjson_warn_unused simdjson_inline error_code tape_builder::parse_document(
-    dom_parser_implementation &dom_parser,
-    dom::document &doc) noexcept {
-  dom_parser.doc = &doc;
-  json_iterator iter(dom_parser, STREAMING ? dom_parser.next_structural_index : 0);
-  tape_builder builder(doc);
-  return iter.walk_document<STREAMING>(builder);
-}
+// Thin, non-templated entry so each architecture's stage2() keeps calling
+// tape_builder::parse_document<STREAMING> unchanged. It chooses the bounds-safe
+// (unpadded) or the regular (padded) tape_builder_impl ONCE per document, so the
+// choice is a compile-time constant inside the walk: the padded path carries no
+// extra branch or load (see tape_builder_impl::visit_string).
+struct tape_builder {
+  template<bool STREAMING>
+  simdjson_warn_unused static simdjson_inline error_code parse_document(
+      dom_parser_implementation &dom_parser, dom::document &doc) noexcept {
+    dom_parser.doc = &doc;
+    json_iterator iter(dom_parser, STREAMING ? dom_parser.next_structural_index : 0);
+    if (dom_parser._unpadded) {
+      tape_builder_impl<true> builder(doc);
+      return iter.walk_document<STREAMING, true>(builder);
+    } else {
+      tape_builder_impl<false> builder(doc);
+      return iter.walk_document<STREAMING, false>(builder);
+    }
+  }
+};

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_primitive(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_primitive(json_iterator &iter, const uint8_t *value) noexcept {
   return iter.visit_root_primitive(*this, value);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_primitive(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_primitive(json_iterator &iter, const uint8_t *value) noexcept {
   return iter.visit_primitive(*this, value);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_empty_object(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_empty_object(json_iterator &iter) noexcept {
   return empty_container(iter, internal::tape_type::START_OBJECT, internal::tape_type::END_OBJECT);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_empty_array(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_empty_array(json_iterator &iter) noexcept {
   return empty_container(iter, internal::tape_type::START_ARRAY, internal::tape_type::END_ARRAY);
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_document_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_document_start(json_iterator &iter) noexcept {
   start_container(iter);
   return SUCCESS;
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_object_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_object_start(json_iterator &iter) noexcept {
   start_container(iter);
   return SUCCESS;
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_array_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_array_start(json_iterator &iter) noexcept {
   start_container(iter);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_object_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_object_end(json_iterator &iter) noexcept {
   return end_container(iter, internal::tape_type::START_OBJECT, internal::tape_type::END_OBJECT);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_array_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_array_end(json_iterator &iter) noexcept {
   return end_container(iter, internal::tape_type::START_ARRAY, internal::tape_type::END_ARRAY);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_document_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_document_end(json_iterator &iter) noexcept {
   constexpr uint32_t start_tape_index = 0;
   tape.append(start_tape_index, internal::tape_type::ROOT);
   tape_writer::write(iter.dom_parser.doc->tape[start_tape_index], next_tape_index(iter), internal::tape_type::ROOT);
   return SUCCESS;
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_key(json_iterator &iter, const uint8_t *key) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_key(json_iterator &iter, const uint8_t *key) noexcept {
   return visit_string(iter, key, true);
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::increment_count(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::increment_count(json_iterator &iter) noexcept {
   iter.dom_parser.open_containers[iter.depth].count++; // we have a key value pair in the object at parser.dom_parser.depth - 1
   return SUCCESS;
 }

-simdjson_inline tape_builder::tape_builder(dom::document &doc) noexcept : tape{doc.tape.get()}, current_string_buf_loc{doc.string_buf.get()} {}
+template <bool UNPADDED>
+simdjson_inline tape_builder_impl<UNPADDED>::tape_builder_impl(dom::document &doc) noexcept : tape{doc.tape.get()}, current_string_buf_loc{doc.string_buf.get()} {}

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_string(json_iterator &iter, const uint8_t *value, bool key) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_string(json_iterator &iter, const uint8_t *value, bool key) noexcept {
   iter.log_value(key ? "key" : "string");
   uint8_t *dst = on_start_string(iter);
-  dst = stringparsing::parse_string(value+1, dst, false); // We do not allow replacement when the escape characters are invalid.
+  // We do not allow replacement when the escape characters are invalid.
+  // UNPADDED is a compile-time constant chosen once per document by
+  // tape_builder::parse_document, so the padded build instantiates only the
+  // plain parse_string call below -- no runtime branch and no flag load.
+  SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+    dst = stringparsing::parse_string_safe(value+1, dst, false, iter.buf + iter.dom_parser.len);
+  } else {
+    dst = stringparsing::parse_string(value+1, dst, false);
+  }
   if (dst == nullptr) {
     iter.log_error("Invalid escape in string");
     return STRING_ERROR;
@@ -34716,27 +44426,48 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_string(json_
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_string(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_string(json_iterator &iter, const uint8_t *value) noexcept {
   return visit_string(iter, value);
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_number(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_number(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("number");
-  error_code err = numberparsing::parse_number(value, tape);
+  const uint8_t *num = value;
+  std::unique_ptr<uint8_t[]> copy{}; // keeps a padded copy of the tail alive when used
+  SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+    // numberparsing reads ahead in 8-byte blocks for floats
+    // (is_made_of_eight_digits_fast reads up to 7 bytes past the digits), so a
+    // number whose digits reach the final bytes of an unpadded buffer would read
+    // past it. *(next_structural) is the offset of the token following this
+    // number, hence an upper bound on where the digits end; when that is within
+    // SIMDJSON_PADDING of the end we parse from a space-padded copy of the tail
+    // (mirroring visit_root_number). This fires only for numbers near the end.
+    if (simdjson_unlikely(*(iter.next_structural) + SIMDJSON_PADDING > iter.dom_parser.len)) {
+      const size_t rl = iter.remaining_len(); // bytes from `value` to the end of the document
+      copy.reset(new (std::nothrow) uint8_t[rl + SIMDJSON_PADDING]);
+      if (copy.get() == nullptr) { return MEMALLOC; }
+      std::memcpy(copy.get(), value, rl);
+      std::memset(copy.get() + rl, ' ', SIMDJSON_PADDING);
+      num = copy.get();
+    }
+  }
+  error_code err = numberparsing::parse_number(num, tape);
   if (simdjson_unlikely(err == BIGINT_ERROR &&
       iter.dom_parser._number_as_string)) {
     // Write big integer to string buffer using the same format as strings.
     // Scan digits the same way parse_number does (skip optional '-', then digits).
-    const uint8_t *p = value;
+    const uint8_t *p = num;
     if (*p == '-') p++;
     while (numberparsing::is_digit(*p)) p++;
     // The digit run must be terminated by a structural or whitespace character; otherwise the
     // token is malformed (e.g. "123456789123456789123x").
     if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
-    size_t len = size_t(p - value);
+    size_t len = size_t(p - num);
     tape.append(current_string_buf_loc - iter.dom_parser.doc->string_buf.get(), internal::tape_type::BIGINT);
     uint8_t *dst = current_string_buf_loc + sizeof(uint32_t);
-    memcpy(dst, value, len);
+    memcpy(dst, num, len);
     dst += len;
     on_end_string(dst);
     return SUCCESS;
@@ -34744,7 +44475,8 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_number(json_
   return err;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_number(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_number(json_iterator &iter, const uint8_t *value) noexcept {
   //
   // We need to make a copy to make sure that the string is space terminated.
   // This is not about padding the input, which should already padded up
@@ -34758,76 +44490,142 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_number(
   // practice unless you are in the strange scenario where you have many JSON
   // documents made of single atoms.
   //
-  std::unique_ptr<uint8_t[]>copy(new (std::nothrow) uint8_t[iter.remaining_len() + SIMDJSON_PADDING]);
+  // In a stream, the input goes on with other documents: copy up to the next
+  // structural only, not to the end of the batch.
+  const size_t len = (std::min)(iter.remaining_len(), size_t(*iter.next_structural) - size_t(*(iter.next_structural - 1)));
+  std::unique_ptr<uint8_t[]>copy(new (std::nothrow) uint8_t[len + SIMDJSON_PADDING]);
   if (copy.get() == nullptr) { return MEMALLOC; }
-  std::memcpy(copy.get(), value, iter.remaining_len());
-  std::memset(copy.get() + iter.remaining_len(), ' ', SIMDJSON_PADDING);
+  std::memcpy(copy.get(), value, len);
+  std::memset(copy.get() + len, ' ', SIMDJSON_PADDING);
   error_code error = visit_number(iter, copy.get());
   return error;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("true");
-  if (!atomparsing::is_valid_true_atom(value)) { return T_ATOM_ERROR; }
+  // The non-length-aware validator reads a fixed 5 bytes; a malformed/truncated
+  // token at the very end of an unpadded buffer would over-read. Use the
+  // length-aware form there (the root variant already does this).
+  const bool ok = UNPADDED ? atomparsing::is_valid_true_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_true_atom(value);
+  if (!ok) { return T_ATOM_ERROR; }
   tape.append(0, internal::tape_type::TRUE_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("true");
   if (!atomparsing::is_valid_true_atom(value, iter.remaining_len())) { return T_ATOM_ERROR; }
   tape.append(0, internal::tape_type::TRUE_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("false");
-  if (!atomparsing::is_valid_false_atom(value)) { return F_ATOM_ERROR; }
+  const bool ok = UNPADDED ? atomparsing::is_valid_false_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_false_atom(value);
+  if (!ok) { return F_ATOM_ERROR; }
   tape.append(0, internal::tape_type::FALSE_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("false");
   if (!atomparsing::is_valid_false_atom(value, iter.remaining_len())) { return F_ATOM_ERROR; }
   tape.append(0, internal::tape_type::FALSE_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("null");
-  if (!atomparsing::is_valid_null_atom(value)) { return N_ATOM_ERROR; }
+  const bool ok = UNPADDED ? atomparsing::is_valid_null_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_null_atom(value);
+  if (!ok) { return N_ATOM_ERROR; }
   tape.append(0, internal::tape_type::NULL_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("null");
   if (!atomparsing::is_valid_null_atom(value, iter.remaining_len())) { return N_ATOM_ERROR; }
   tape.append(0, internal::tape_type::NULL_VALUE);
   return SUCCESS;
 }

+#if SIMDJSON_ENABLE_NAN_INF
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept {
+  iter.log_value("nan");
+  // For unpadded input use the length-aware validator so the 'infinity'-style
+  // 8-byte compare cannot read past the buffer on a malformed token at the end.
+  const bool ok = UNPADDED ? atomparsing::is_valid_nan_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_nan_atom(value);
+  if (!ok) { return errc; }
+  tape.append_double(std::numeric_limits<double>::quiet_NaN());
+  return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept {
+  iter.log_value("nan");
+  if (!atomparsing::is_valid_nan_atom(value, iter.remaining_len())) { return errc; }
+  tape.append_double(std::numeric_limits<double>::quiet_NaN());
+  return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_inf_atom(json_iterator &iter, const uint8_t *value) noexcept {
+  iter.log_value("inf");
+  // Because 'inf' is an extension, non a canonical atom, a tape error should be returned on failure.
+  // For unpadded input use the length-aware validator so the 'infinity' 8-byte
+  // compare cannot read past the buffer on a malformed token at the end.
+  const bool ok = UNPADDED ? atomparsing::is_valid_inf_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_inf_atom(value);
+  if (!ok) { return TAPE_ERROR; }
+  tape.append_double(std::numeric_limits<double>::infinity());
+  return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_inf_atom(json_iterator &iter, const uint8_t *value) noexcept {
+  iter.log_value("inf");
+  // Because 'inf' is an extension, non a canonical atom, a tape error should be returned on failure
+  if (!atomparsing::is_valid_inf_atom(value, iter.remaining_len())) { return TAPE_ERROR; }
+  tape.append_double(std::numeric_limits<double>::infinity());
+  return SUCCESS;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
 // private:

-simdjson_inline uint32_t tape_builder::next_tape_index(json_iterator &iter) const noexcept {
+template <bool UNPADDED>
+simdjson_inline uint32_t tape_builder_impl<UNPADDED>::next_tape_index(json_iterator &iter) const noexcept {
   return uint32_t(tape.next_tape_loc - iter.dom_parser.doc->tape.get());
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
   auto start_index = next_tape_index(iter);
   tape.append(start_index+2, start);
   tape.append(start_index, end);
   return SUCCESS;
 }

-simdjson_inline void tape_builder::start_container(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_inline void tape_builder_impl<UNPADDED>::start_container(json_iterator &iter) noexcept {
   iter.dom_parser.open_containers[iter.depth].tape_index = next_tape_index(iter);
   iter.dom_parser.open_containers[iter.depth].count = 0;
   tape.skip(); // We don't actually *write* the start element until the end.
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
   // Write the ending tape element, pointing at the start location
   const uint32_t start_tape_index = iter.dom_parser.open_containers[iter.depth].tape_index;
   tape.append(start_tape_index, end);
@@ -34840,13 +44638,15 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::end_container(json
   return SUCCESS;
 }

-simdjson_inline uint8_t *tape_builder::on_start_string(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_inline uint8_t *tape_builder_impl<UNPADDED>::on_start_string(json_iterator &iter) noexcept {
   // we advance the point, accounting for the fact that we have a NULL termination
   tape.append(current_string_buf_loc - iter.dom_parser.doc->string_buf.get(), internal::tape_type::STRING);
   return current_string_buf_loc + sizeof(uint32_t);
 }

-simdjson_inline void tape_builder::on_end_string(uint8_t *dst) noexcept {
+template <bool UNPADDED>
+simdjson_inline void tape_builder_impl<UNPADDED>::on_end_string(uint8_t *dst) noexcept {
   uint32_t str_length = uint32_t(dst - (current_string_buf_loc + sizeof(uint32_t)));
   // TODO check for overflow in case someone has a crazy string (>=4GB?)
   // But only add the overflow check when the document itself exceeds 4GB
@@ -35161,16 +44961,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
 }
 #endif

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
-                                uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
-  return _addcarry_u64(0, value1, value2,
-                       reinterpret_cast<unsigned __int64 *>(result));
-#else
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-#endif
-}

 } // unnamed namespace
 } // namespace westmere
@@ -35749,16 +45539,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
 }
 #endif

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
-                                uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
-  return _addcarry_u64(0, value1, value2,
-                       reinterpret_cast<unsigned __int64 *>(result));
-#else
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-#endif
-}

 } // unnamed namespace
 } // namespace westmere
@@ -36372,6 +46152,9 @@ namespace atomparsing {
 // to the compile-time constant 1936482662.
 simdjson_inline uint32_t string_to_uint32(const char* str) { uint32_t val; std::memcpy(&val, str, sizeof(uint32_t)); return val; }

+// Acts on the same principle as string_to_uint32, but on an 8-byte block of memory
+simdjson_inline uint64_t string_to_uint64(const char* str) { uint64_t val; std::memcpy(&val, str, sizeof(uint64_t)); return val; }
+

 // Again in str4ncmp we use a memcpy to avoid undefined behavior. The memcpy may appear expensive.
 // Yet all decent optimizing compilers will compile memcpy to a single instruction, just about.
@@ -36383,6 +46166,28 @@ simdjson_inline uint32_t str4ncmp(const uint8_t *src, const char* atom) {
   return srcval ^ string_to_uint32(atom);
 }

+// Checks that the first 8 characters of the input string match the given atom in a case-insensitive manner.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint64_t str8ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+  uint64_t srcval; // we want to avoid unaligned 32-bit loads (undefined in C/C++)
+  static_assert(sizeof(uint64_t) <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be larger than 8 bytes");
+  std::memcpy(&srcval, src, sizeof(uint64_t));
+
+  return (srcval | 0x2020202020202020ull) ^ string_to_uint64(atom);
+}
+
+// Checks that the first 3 characters of 'src' match 'atom' in a case-insensitive way.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint32_t str3ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+  return ((src[0] | 0x20) ^ atom[0]) //
+       | ((src[1] | 0x20) ^ atom[1]) //
+       | ((src[2] | 0x20) ^ atom[2]);
+}
+
 simdjson_warn_unused
 simdjson_inline bool is_valid_true_atom(const uint8_t *src) {
   return (str4ncmp(src, "true") | jsoncharutils::is_not_structural_or_whitespace(src[4])) == 0;
@@ -36419,6 +46224,71 @@ simdjson_inline bool is_valid_null_atom(const uint8_t *src, size_t len) {
   else { return false; }
 }

+#if SIMDJSON_ENABLE_NAN_INF
+// "nan" is 3 bytes; we check characters and then verify the next
+// character is structural or whitespace. We accept both "nan" and "NaN".
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src) {
+  return (str3ncmp_case_insensitive(src, "nan")
+        | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+}
+
+// checks that the next four characters of a string are 'nan"', where the 'nan'
+// is checked in a case-insensitive way.
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_in_string(const uint8_t *src) {
+  return (str3ncmp_case_insensitive(src, "nan") | (src[3] ^ '"')) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src, size_t len) {
+  if (len > 3) { return is_valid_nan_atom(src); }
+  if (len == 3) { return str3ncmp_case_insensitive(src, "nan") == 0; }
+  return false;
+}
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src) {
+  bool is_short_inf = (str3ncmp_case_insensitive(src, "inf")
+                     | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+  if(is_short_inf) return true;
+
+  // Check for 'infinity' (any capitalization)
+  return (str8ncmp_case_insensitive(src, "infinity") | jsoncharutils::is_not_structural_or_whitespace(src[8])) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_in_string(const uint8_t *src) {
+  bool is_short_inf = (str3ncmp_case_insensitive(src, "inf") | (src[3] ^ '"')) == 0;
+  if(is_short_inf) return true;
+
+  return (str8ncmp_case_insensitive(src, "infinity") | (src[8] ^ '"')) == 0;
+}
+
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src, size_t len) {
+  if (len > 8) { return is_valid_inf_atom(src); }
+  if (len == 8 && str8ncmp_case_insensitive(src, "infinity") == 0) {
+    return true;
+  }
+  if (len > 3) {
+    return (str3ncmp_case_insensitive(src, "inf")
+          | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+  }
+  if (len == 3) { return str3ncmp_case_insensitive(src, "inf") == 0; }
+  return false;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
 } // namespace atomparsing
 } // unnamed namespace
 } // namespace westmere
@@ -36509,7 +46379,7 @@ inline simdjson_warn_unused error_code dom_parser_implementation::set_capacity(s
 }

 inline simdjson_warn_unused error_code dom_parser_implementation::set_max_depth(size_t max_depth) noexcept {
-  if(max_depth == 0 || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
+  if(max_depth == 0 || max_depth > SIMDJSON_MAX_DEPTH || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
   // Stage 2 stacks
   open_containers.reset(new (std::nothrow) open_container[max_depth]);
   is_array.reset(new (std::nothrow) bool[max_depth]);
@@ -36688,6 +46558,7 @@ protected:
 /* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_NUMBERPARSING_H */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/base.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/jsoncharutils.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/atomparsing.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/internal/numberparsing_tables.h" */
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */

@@ -36727,6 +46598,18 @@ simdjson_inline double to_double(uint64_t mantissa, uint64_t real_exponent, bool
     return d;
 }

+// Convert a mantissa, an exponent and a sign bit into an ieee32 float (binary32).
+// The real_exponent needs to be in [0, 254] (technically real_exponent = 255 would be acceptable).
+// The mantissa should be in [0,1<<24). The bit at index (1U << 23) will be zeroed.
+simdjson_inline float to_float(uint32_t mantissa, uint32_t real_exponent, bool negative) {
+    float f;
+    mantissa &= ~(uint32_t(1) << 23);
+    mantissa |= real_exponent << 23;
+    mantissa |= ((static_cast<uint32_t>(negative)) << 31);
+    std::memcpy(&f, &mantissa, sizeof(f));
+    return f;
+}
+
 // Attempts to compute i * 10^(power) exactly; and if "negative" is
 // true, negate the result.
 // This function will only work in some cases, when it does not work, success is
@@ -36983,6 +46866,219 @@ simdjson_inline bool compute_float_64(int64_t power, uint64_t i, bool negative,
   return true;
 }

+// Attempts to compute i * 10^(power) as a binary32 (float) value; and if
+// "negative" is true, negate the result.
+//
+// This is the same Eisel-Lemire algorithm as compute_float_64, retargeted at
+// binary32. It is adapted from fast_float
+// (https://github.com/fastfloat/fast_float), where the routine is written once
+// and instantiated for each binary format. We only need 24 bits of mantissa
+// instead of 53, so we ask the truncated product for 24 + 2 = 26 accurate bits
+// (plus one bit that may be lost to the "upperbit" shift), and we shift the
+// product right by (upperbit + 64 - 23 - 3) instead of (upperbit + 64 - 52 - 3).
+//
+// The power-of-five table is shared with the binary64 code: the useful range
+// for binary32, [smallest_power_binary32, largest_power_binary32], is a strict
+// subset of [smallest_power, largest_power].
+//
+// The function returns false when the result would be infinite: simdjson
+// refuses to parse infinite values, so the caller reports an error. Unlike
+// compute_float_64, a false return never means "try harder": the accuracy of
+// the two-limb product is guaranteed by Noble Mushtak and Daniel Lemire,
+// "Fast Number Parsing Without Fallback" (https://arxiv.org/abs/2212.06644).
+// The caller must still fall back to a slow path when the 64-bit mantissa i
+// was truncated (more than 19 significant digits).
+//
+// We assume that power is in the [smallest_power, largest_power] interval: the
+// caller is responsible for this check.
+simdjson_inline bool compute_float_32(int64_t power, uint64_t i, bool negative, float &d) {
+  // Powers of ten that are exactly representable as binary32 values: 10^k is
+  // exact as long as k <= 10 (5^10 = 9765625 < 2**24 while 5^11 > 2**24).
+  static constexpr float power_of_ten_float[] = {
+      1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+      1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+  // The range of powers of ten that a non-zero, finite binary32 value can be
+  // built from. Anything smaller rounds to zero, anything larger is infinite.
+  // These are the smallest_power_of_ten()/largest_power_of_ten() constants that
+  // fast_float uses for binary32.
+  constexpr int smallest_power_binary32 = -65;
+  constexpr int largest_power_binary32 = 38;
+
+  // We start with the fast path described in
+  // Clinger WD. How to read floating point numbers accurately.
+  // ACM SIGPLAN Notices. 1990
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+  // We cannot be certain that x/y is rounded to nearest.
+  if (0 <= power && power <= 10 && i <= 16777215)
+#else
+  if (-10 <= power && power <= 10 && i <= 16777215)
+#endif
+  {
+    // Convert the integer into a float. This is lossless since
+    // 0 <= i <= 2^24 - 1.
+    d = float(i);
+    // Both d and the power of ten are exactly representable as binary32
+    // values, so the product (or quotient) is correctly rounded.
+    if (power < 0) {
+      d = d / power_of_ten_float[-power];
+    } else {
+      d = d * power_of_ten_float[power];
+    }
+    if (negative) {
+      d = -d;
+    }
+    return true;
+  }
+
+  // The fast path has failed, so we fall back on the Eisel-Lemire algorithm.
+  // It needs i > 0 (so that the leading bit of i can be normalized), so we
+  // handle i == 0 separately. We also handle the powers of ten that are so
+  // small (or so large) that the answer is zero (or infinite) whatever the
+  // mantissa is.
+  if (i == 0 || power < smallest_power_binary32) {
+    d = negative ? -0.0f : 0.0f;
+    return true;
+  }
+  if (power > largest_power_binary32) {
+    // We have, for sure, an infinite value.
+    return false;
+  }
+
+  // The exponent is 128 + 63 + power + floor(log(5**power)/log(2)).
+  // The 128 comes from the ieee32 standard (the minimal exponent is -127).
+  // The 63 comes from the fact that we use a 64-bit word.
+  // See compute_float_64 for a discussion of the magical 152170 + 65536.
+  int64_t exponent = (((152170 + 65536) * power) >> 16) + 128 + 63;
+
+  // We want the most significant bit of i to be 1. Shift if needed.
+  int lz = leading_zeroes(i);
+  i <<= lz;
+
+  // We want the most significant 64 bits of the product i * 5**power. It is
+  // safe to index the table because
+  // smallest_power <= smallest_power_binary32 <= power
+  //               <= largest_power_binary32 <= largest_power.
+  const uint32_t index = 2 * uint32_t(power - simdjson::internal::smallest_power);
+#if SIMDJSON_STATIC_REFLECTION
+  simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index]);
+#else
+  simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index]);
+#endif
+
+  // Unless the least significant 38 bits of the high (64-bit) part of the full
+  // product are all 1s, then we know that the most significant 26 bits are
+  // exact and no further work is needed. Having 26 bits is necessary because
+  // we need 24 bits for the mantissa but we have to have one rounding bit and
+  // we can waste a bit if the most significant bit of the product is zero.
+  // (For binary64, the same reasoning gives 55 bits and a 9-bit mask.)
+  if((firstproduct.high & 0x3FFFFFFFFF) == 0x3FFFFFFFFF) {
+    // The truncated multiplication was not accurate enough; use the next 64
+    // bits of the power of five to refine it. See compute_float_64 for a
+    // detailed discussion.
+#if SIMDJSON_STATIC_REFLECTION
+    simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index + 1]);
+#else
+    simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index + 1]);
+#endif
+    firstproduct.low += secondproduct.high;
+    if(secondproduct.high > firstproduct.low) { firstproduct.high++; }
+  }
+  uint64_t lower = firstproduct.low;
+  uint64_t upper = firstproduct.high;
+  // The final mantissa should be 24 bits with a leading 1.
+  // We shift it so that it occupies 25 bits with a leading 1.
+  ///////
+  uint64_t upperbit = upper >> 63;
+  uint64_t mantissa = upper >> (upperbit + 38); // 38 == 64 - 23 - 3
+  lz += int(1 ^ upperbit);
+
+  // Here we have mantissa < (1<<25).
+  int64_t real_exponent = exponent - lz;
+  if (simdjson_unlikely(real_exponent <= 0)) { // we have a subnormal?
+    // Here we have that real_exponent <= 0 so -real_exponent >= 0
+    if(-real_exponent + 1 >= 64) { // if we have more than 64 bits below the minimum exponent, you have a zero for sure.
+      d = negative ? -0.0f : 0.0f;
+      return true;
+    }
+    // next line is safe because -real_exponent + 1 < 64
+    mantissa >>= -real_exponent + 1;
+    // Thankfully, we can't have both "round-to-even" and subnormals because
+    // "round-to-even" only occurs for powers close to 0.
+    mantissa += (mantissa & 1); // round up
+    mantissa >>= 1;
+    // As in compute_float_64, rounding up may take us out of the subnormal
+    // range, so we can only decide after rounding.
+    real_exponent = (mantissa < (uint64_t(1) << 23)) ? 0 : 1;
+    d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+    return true;
+  }
+  // We have to round to even. The "to even" part is only a problem when we are
+  // right in between two floats, which we guard against. The bounds on the
+  // power of ten are those of fast_float's
+  // min_exponent_round_to_even()/max_exponent_round_to_even() for binary32:
+  // when q >= 0, (2m+1) must be divisible by 5^q with 5^q <= 2^25, so q <= 10;
+  // when q < 0, we need 2^24 x 5^{-q} < 2^{64}, so q >= -17.
+  if (simdjson_unlikely((lower <= 1) && (power >= -17) && (power <= 10) && ((mantissa & 3) == 1))) {
+    if((mantissa << (upperbit + 38)) == upper) {
+      mantissa &= ~uint64_t(1);   // flip it so that we do not round up
+    }
+  }
+
+  mantissa += mantissa & 1;
+  mantissa >>= 1;
+
+  // Here we have mantissa < (1<<24), unless there was an overflow
+  if (mantissa >= (uint64_t(1) << 24)) {
+    mantissa = (uint64_t(1) << 23);
+    real_exponent++;
+  }
+  mantissa &= ~(uint64_t(1) << 23);
+  // we have to check that real_exponent is in range, otherwise we bail out
+  if (simdjson_unlikely(real_exponent > 254)) {
+    // We have an infinite value!!! We could actually throw an error here if we could.
+    return false;
+  }
+  d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+  return true;
+}
+
+#if SIMDJSON_ENABLE_NAN_INF
+// Parses a nan or infinity. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, double& d) noexcept {
+  if (atomparsing::is_valid_inf_atom(src)) {
+    double inf = std::numeric_limits<double>::infinity();
+    d = negative ? -inf : inf;
+    return true;
+  }
+
+  if (atomparsing::is_valid_nan_atom(src)) {
+    d = std::numeric_limits<double>::quiet_NaN();
+    return true;
+  }
+
+  return false;
+}
+
+// Parses a nan or infinity as a binary32 value. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, float& d) noexcept {
+  if (atomparsing::is_valid_inf_atom(src)) {
+    float inf = std::numeric_limits<float>::infinity();
+    d = negative ? -inf : inf;
+    return true;
+  }
+
+  if (atomparsing::is_valid_nan_atom(src)) {
+    d = std::numeric_limits<float>::quiet_NaN();
+    return true;
+  }
+
+  return false;
+}
+#endif
+
 // We call a fallback floating-point parser that might be slow. Note
 // it will accept JSON numbers, but the JSON spec. is more restrictive so
 // before you call parse_float_fallback, you need to have validated the input
@@ -37020,6 +47116,16 @@ static bool parse_float_fallback(const uint8_t *ptr, const uint8_t *end_ptr, dou
   return !(*outDouble > (std::numeric_limits<double>::max)() || *outDouble < std::numeric_limits<double>::lowest());
 }

+// Same as parse_float_fallback, but for binary32 (float) values. Going through
+// the binary64 fallback and then rounding to binary32 would be subject to
+// double rounding, so we run the fallback algorithm directly on binary32.
+static bool parse_float_fallback(const uint8_t *ptr, float *outFloat) {
+  *outFloat = simdjson::internal::from_chars_float(reinterpret_cast<const char *>(ptr));
+  // We do not accept infinite values. See the binary64 version above for why we
+  // do not use std::isfinite.
+  return !(*outFloat > (std::numeric_limits<float>::max)() || *outFloat < std::numeric_limits<float>::lowest());
+}
+
 // check quickly whether the next 8 chars are made of digits
 // at a glance, it looks better than Mula's
 // http://0x80.pl/articles/swar-digits-validate.html
@@ -37038,6 +47144,28 @@ simdjson_inline bool is_made_of_eight_digits_fast(const uint8_t *chars) {
           0x3333333333333333);
 }

+// The same idea for four characters. A block of eight only bites when eight
+// digits are left, so a run of 4 to 7 digits used to fall back to one digit at a
+// time; taking four of them at once is what makes the tail of a long fraction
+// cheap. Adapted from fast_float, credit @aqrit.
+simdjson_inline bool is_made_of_four_digits_fast(const uint8_t *chars) {
+  uint32_t val;
+  // Reads up to 3 bytes beyond the digits, which SIMDJSON_PADDING covers.
+  static_assert(3 <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be bigger than 3");
+  std::memcpy(&val, chars, 4);
+  return !(((val + 0x46464646) | (val - 0x30303030)) & 0x80808080);
+}
+
+// Only call this when is_made_of_four_digits_fast() says the four characters are
+// digits. Little-endian: guarded by SIMDJSON_SWAR_NUMBER_PARSING.
+simdjson_inline uint32_t parse_four_digits_unrolled(const uint8_t *chars) {
+  uint32_t val;
+  std::memcpy(&val, chars, 4);
+  val -= 0x30303030;
+  val = (val * 10) + (val >> 8);
+  return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
 template<typename I>
 SIMDJSON_NO_SANITIZE_UNDEFINED // We deliberately allow overflow here and check later
 simdjson_inline bool parse_digit(const uint8_t c, I &i) {
@@ -37054,26 +47182,73 @@ simdjson_inline bool is_digit(const uint8_t c) {
   return static_cast<uint8_t>(c - '0') <= 9;
 }

-simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
-  // we continue with the fiction that we have an integer. If the
-  // floating point number is representable as x * 10^z for some integer
-  // z that fits in 53 bits, then we will be able to convert back the
-  // the integer into a float in a lossless manner.
-  const uint8_t *const first_after_period = p;
+// Consumes a run of digits into i, eight at a time while eight are available,
+// then four, then one at a time. Overflow is deliberate: the caller counts the
+// digits and falls back when there are too many for a 64-bit mantissa.
+//
+// Long runs of digits are what a float-heavy document is made of: the fraction
+// of a binary64 printed to full precision is 15 to 17 digits, and reading those
+// one at a time is the single largest cost in parse_double(). Integer parts are
+// left alone: they are usually a handful of digits, and there the check for a
+// block of eight is wasted work.
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_fraction_digits(const uint8_t *&p, uint64_t &i) {
+#ifdef SIMDJSON_SWAR_NUMBER_PARSING
+#if SIMDJSON_SWAR_NUMBER_PARSING
+  while (is_made_of_eight_digits_fast(p)) {
+    i = i * 100000000 + parse_eight_digits_unrolled(p);
+    p += 8;
+  }
+  // A 4 to 7 digit remainder is the common case once the blocks of eight are
+  // gone: 15 fraction digits leave 7, and 17 leave 1 after two blocks.
+  if (is_made_of_four_digits_fast(p)) {
+    i = i * 10000 + parse_four_digits_unrolled(p);
+    p += 4;
+  }
+#endif // SIMDJSON_SWAR_NUMBER_PARSING
+#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
+  while (parse_digit(*p, i)) { p++; }
+}

+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_float_integer_digits(const uint8_t *&p, uint64_t &i,
+                                                bool &leading_zero) {
+  if (!parse_digit(*p, i)) { leading_zero = true; return; }
+  p++;
+  leading_zero = (i == 0);
+  while (parse_digit(*p, i)) { p++; }
+}
+
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_integer_digits(const uint8_t *&p, uint64_t &i) {
 #ifdef SIMDJSON_SWAR_NUMBER_PARSING
 #if SIMDJSON_SWAR_NUMBER_PARSING
-  // this helps if we have lots of decimals!
-  // this turns out to be frequent enough.
+  // Identifiers, timestamps and counters often have eight digits or more.
+  // Prior related work: jsonifier parses integers as eight-digit SWAR words
+  // (str_to_i.hpp, https://github.com/nihilai-collective/Jsonifier). This
+  // takes one such word with parse_eight_digits_unrolled, the routine
+  // simdjson uses for long fractions.
   if (is_made_of_eight_digits_fast(p)) {
     i = i * 100000000 + parse_eight_digits_unrolled(p);
     p += 8;
   }
+  const uint8_t *const swar_end = p + 8;
+  while (p < swar_end && is_made_of_four_digits_fast(p)) {
+    i = i * 10000 + parse_four_digits_unrolled(p);
+    p += 4;
+  }
 #endif // SIMDJSON_SWAR_NUMBER_PARSING
 #endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
-  // Unrolling the first digit makes a small difference on some implementations (e.g. westmere)
-  if (parse_digit(*p, i)) { ++p; }
   while (parse_digit(*p, i)) { p++; }
+}
+
+simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
+  // we continue with the fiction that we have an integer. If the
+  // floating point number is representable as x * 10^z for some integer
+  // z that fits in 53 bits, then we will be able to convert back the
+  // the integer into a float in a lossless manner.
+  const uint8_t *const first_after_period = p;
+  parse_fraction_digits(p, i);
   exponent = first_after_period - p;
   // Decimal without digits (123.) is illegal
   if (exponent == 0) {
@@ -37162,7 +47337,7 @@ simdjson_inline size_t significant_digits(const uint8_t * start_digits, size_t d
 } // unnamed namespace

 /** @private */
-static error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
+inline error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
   if (parse_float_fallback(src, answer)) {
     return SUCCESS;
   }
@@ -37245,15 +47420,17 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
   return SUCCESS;              // always succeeds
 }

-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept  { return false; }
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept  { return false; }
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept { return number_type::signed_integer; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline bool is_negative(const uint8_t *) noexcept  { return false; }
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t *) noexcept  { return false; }
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t *) noexcept { return number_type::signed_integer; }
 #else

 // parse the number at src
@@ -37284,7 +47461,21 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
   size_t digit_count = size_t(p - start_digits);
-  if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) { return INVALID_NUMBER(src); }
+  if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // By this point, we know that our input does not begin with a digit. We will attempt
+    // to handle NaN/Infinity.
+
+    double d;
+    if (compute_nan_inf(p, negative, d)) {
+      writer.append_double(d);
+      return SUCCESS;
+    }
+#endif
+
+    return INVALID_NUMBER(src);
+  }

   //
   // Handle floats if there is a . or e (or both)
@@ -37333,7 +47524,7 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
     // - Therefore, if the number is positive and lower than that, it's overflow.
     // - The value we are looking at is less than or equal to INT64_MAX.
     //
-    }  else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return INVALID_NUMBER(src); }
+    }  else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return BIGINT_NUMBER(src); }
   }

   // Write unsigned if it does not fit in a signed integer.
@@ -37432,7 +47623,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const u
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -37530,7 +47721,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_stri
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -37585,7 +47776,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uin
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -37671,7 +47862,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = src;
   uint64_t i = 0;
-  while (parse_digit(*src, i)) { src++; }
+  parse_integer_digits(src, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -37711,9 +47902,247 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   //
   uint64_t i = 0;
   const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no loading digits, the number may be nan or infinity.
+    // Attempt to compute those, and return on success.
+    double d;
+    if (compute_nan_inf(p, negative, d)) { return d; }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = p-start_digits > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if (*p == 'e' || *p == 'E') {
+    p++;
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while (parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  double d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_64(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
+// Parse a JSON number into a binary32 (float) value.
+//
+// This mirrors parse_double, but it rounds to binary32 directly instead of
+// rounding to binary64 and then to binary32: the latter is subject to double
+// rounding and would not always produce the nearest float.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * src) noexcept {
+  //
+  // Check for minus sign
+  //
+  bool negative = (*src == '-');
+  src += uint8_t(negative);
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, the number may be nan or infinity.
+    // Attempt to compute those, and return on success.
+    float f;
+    if (compute_nan_inf(p, negative, f)) { return f; }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = p-start_digits > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if (*p == 'e' || *p == 'E') {
+    p++;
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while (parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  float d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_32(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
+simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
+  return (*src == '-');
+}
+
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept {
+  bool negative = (*src == '-');
+  src += uint8_t(negative);
+  const uint8_t *p = src;
+  while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
+  if ( p == src ) { return NUMBER_ERROR; }
+  if (jsoncharutils::is_structural_or_whitespace(*p)) { return true; }
+  return false;
+}
+
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept {
+  bool negative = (*src == '-');
+  src += uint8_t(negative);
+  const uint8_t *p = src;
+  while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
+  size_t digit_count = size_t(p - src);
+  if ( p == src ) { return NUMBER_ERROR; }
+  if (jsoncharutils::is_structural_or_whitespace(*p)) {
+    static const uint8_t * smaller_big_integer = reinterpret_cast<const uint8_t *>("9223372036854775808");
+    // We have an integer.
+    if(simdjson_unlikely(digit_count > 20)) {
+      return number_type::big_integer;
+    }
+    // If the number is negative and valid, it must be a signed integer.
+    if(negative) {
+      if (simdjson_unlikely(digit_count > 19)) return number_type::big_integer;
+      if (simdjson_unlikely(digit_count == 19 && memcmp(src, smaller_big_integer, 19) > 0)) {
+        return number_type::big_integer;
+      }
+#if SIMDJSON_MINUS_ZERO_AS_FLOAT
+      if(digit_count == 1 && src[0] == '0') {
+        // We have to write -0.0 instead of 0
+        return number_type::floating_point_number;
+      }
+#endif
+      return number_type::signed_integer;
+    }
+    // Let us check if we have a big integer (>=2**64).
+    static const uint8_t * two_to_sixtyfour = reinterpret_cast<const uint8_t *>("18446744073709551616");
+    if((digit_count > 20) || (digit_count == 20 && memcmp(src, two_to_sixtyfour, 20) >= 0)) {
+      return number_type::big_integer;
+    }
+    // The number is positive and smaller than 18446744073709551616 (or 2**64).
+    // We want values larger or equal to 9223372036854775808 to be unsigned
+    // integers, and the other values to be signed integers.
+    if((digit_count == 20) || (digit_count >= 19 && memcmp(src, smaller_big_integer, 19) >= 0)) {
+      return number_type::unsigned_integer;
+    }
+    return number_type::signed_integer;
+  }
+  // Hopefully, we have 'e' or 'E' or '.'.
+  return number_type::floating_point_number;
+}
+
+// Never read at src_end or beyond
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * src, const uint8_t * const src_end) noexcept {
+  if(src == src_end) { return NUMBER_ERROR; }
+  //
+  // Check for minus sign
+  //
+  bool negative = (*src == '-');
+  src += uint8_t(negative);
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  if(p == src_end) { return NUMBER_ERROR; }
   p += parse_digit(*p, i);
   bool leading_zero = (i == 0);
-  while (parse_digit(*p, i)) { p++; }
+  while ((p != src_end) && parse_digit(*p, i)) { p++; }
   // no integer digits, or 0123 (zero must be solo)
   if ( p == src ) { return INCORRECT_TYPE; }
   if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
@@ -37723,12 +48152,104 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   //
   int64_t exponent = 0;
   bool overflow;
-  if (simdjson_likely(*p == '.')) {
+  if (simdjson_likely((p != src_end) && (*p == '.'))) {
     p++;
     const uint8_t *start_decimal_digits = p;
-    if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
+    if ((p == src_end) || !parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
     p++;
-    while (parse_digit(*p, i)) { p++; }
+    while ((p != src_end) && parse_digit(*p, i)) { p++; }
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = start_digits-src > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if ((p != src_end) && (*p == 'e' || *p == 'E')) {
+    p++;
+    if(p == src_end) { return NUMBER_ERROR; }
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while ((p != src_end) && parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if ((p != src_end) && jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  double d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_64(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), src_end, &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * src) noexcept {
+  //
+  // Check for minus sign
+  //
+  bool negative = (*(src + 1) == '-');
+  src += uint8_t(negative) + 1;
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, attempt to parse numbers that are either
+    // NaN or Infinity
+    if (atomparsing::is_valid_inf_in_string(src)) {
+      double inf = std::numeric_limits<double>::infinity();
+      return negative ? -inf : inf;
+    }
+
+    if (atomparsing::is_valid_nan_in_string(src)) {
+      return std::numeric_limits<double>::quiet_NaN();
+    }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
     exponent = -(p - start_decimal_digits);

     // Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -37760,7 +48281,7 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
     exponent += exp_neg ? 0-exp : exp;
   }

-  if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+  if (*p != '"') { return NUMBER_ERROR; }

   overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;

@@ -37777,163 +48298,39 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   return d;
 }

-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
-  return (*src == '-');
-}
-
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept {
-  bool negative = (*src == '-');
-  src += uint8_t(negative);
-  const uint8_t *p = src;
-  while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
-  if ( p == src ) { return NUMBER_ERROR; }
-  if (jsoncharutils::is_structural_or_whitespace(*p)) { return true; }
-  return false;
-}
-
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept {
-  bool negative = (*src == '-');
-  src += uint8_t(negative);
-  const uint8_t *p = src;
-  while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
-  size_t digit_count = size_t(p - src);
-  if ( p == src ) { return NUMBER_ERROR; }
-  if (jsoncharutils::is_structural_or_whitespace(*p)) {
-    static const uint8_t * smaller_big_integer = reinterpret_cast<const uint8_t *>("9223372036854775808");
-    // We have an integer.
-    if(simdjson_unlikely(digit_count > 20)) {
-      return number_type::big_integer;
-    }
-    // If the number is negative and valid, it must be a signed integer.
-    if(negative) {
-      if (simdjson_unlikely(digit_count > 19)) return number_type::big_integer;
-      if (simdjson_unlikely(digit_count == 19 && memcmp(src, smaller_big_integer, 19) > 0)) {
-        return number_type::big_integer;
-      }
-#if SIMDJSON_MINUS_ZERO_AS_FLOAT
-      if(digit_count == 1 && src[0] == '0') {
-        // We have to write -0.0 instead of 0
-        return number_type::floating_point_number;
-      }
-#endif
-      return number_type::signed_integer;
-    }
-    // Let us check if we have a big integer (>=2**64).
-    static const uint8_t * two_to_sixtyfour = reinterpret_cast<const uint8_t *>("18446744073709551616");
-    if((digit_count > 20) || (digit_count == 20 && memcmp(src, two_to_sixtyfour, 20) >= 0)) {
-      return number_type::big_integer;
-    }
-    // The number is positive and smaller than 18446744073709551616 (or 2**64).
-    // We want values larger or equal to 9223372036854775808 to be unsigned
-    // integers, and the other values to be signed integers.
-    if((digit_count == 20) || (digit_count >= 19 && memcmp(src, smaller_big_integer, 19) >= 0)) {
-      return number_type::unsigned_integer;
-    }
-    return number_type::signed_integer;
-  }
-  // Hopefully, we have 'e' or 'E' or '.'.
-  return number_type::floating_point_number;
-}
-
-// Never read at src_end or beyond
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * src, const uint8_t * const src_end) noexcept {
-  if(src == src_end) { return NUMBER_ERROR; }
+// Parse a JSON number held inside a JSON string into a binary32 (float) value.
+// See parse_float for why we do not simply round parse_double_in_string.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * src) noexcept {
   //
   // Check for minus sign
   //
-  bool negative = (*src == '-');
-  src += uint8_t(negative);
+  bool negative = (*(src + 1) == '-');
+  src += uint8_t(negative) + 1;

   //
   // Parse the integer part.
   //
   uint64_t i = 0;
   const uint8_t *p = src;
-  if(p == src_end) { return NUMBER_ERROR; }
-  p += parse_digit(*p, i);
-  bool leading_zero = (i == 0);
-  while ((p != src_end) && parse_digit(*p, i)) { p++; }
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
   // no integer digits, or 0123 (zero must be solo)
-  if ( p == src ) { return INCORRECT_TYPE; }
-  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
-
-  //
-  // Parse the decimal part.
-  //
-  int64_t exponent = 0;
-  bool overflow;
-  if (simdjson_likely((p != src_end) && (*p == '.'))) {
-    p++;
-    const uint8_t *start_decimal_digits = p;
-    if ((p == src_end) || !parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
-    p++;
-    while ((p != src_end) && parse_digit(*p, i)) { p++; }
-    exponent = -(p - start_decimal_digits);
-
-    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
-    overflow = p-src-1 > 19;
-    if (simdjson_unlikely(overflow && leading_zero)) {
-      // Skip leading 0.00000 and see if it still overflows
-      const uint8_t *start_digits = src + 2;
-      while (*start_digits == '0') { start_digits++; }
-      overflow = start_digits-src > 19;
+  if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, attempt to parse numbers that are either
+    // NaN or Infinity
+    if (atomparsing::is_valid_inf_in_string(src)) {
+      float inf = std::numeric_limits<float>::infinity();
+      return negative ? -inf : inf;
     }
-  } else {
-    overflow = p-src > 19;
-  }
-
-  //
-  // Parse the exponent
-  //
-  if ((p != src_end) && (*p == 'e' || *p == 'E')) {
-    p++;
-    if(p == src_end) { return NUMBER_ERROR; }
-    bool exp_neg = *p == '-';
-    p += exp_neg || *p == '+';

-    uint64_t exp = 0;
-    const uint8_t *start_exp_digits = p;
-    while ((p != src_end) && parse_digit(*p, exp)) { p++; }
-    // no exp digits, or 20+ exp digits
-    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
-
-    exponent += exp_neg ? 0-exp : exp;
-  }
-
-  if ((p != src_end) && jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
-
-  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+    if (atomparsing::is_valid_nan_in_string(src)) {
+      return std::numeric_limits<float>::quiet_NaN();
+    }
+#endif

-  //
-  // Assemble (or slow-parse) the float
-  //
-  double d;
-  if (simdjson_likely(!overflow)) {
-    if (compute_float_64(exponent, i, negative, d)) { return d; }
-  }
-  if (!parse_float_fallback(src - uint8_t(negative), src_end, &d)) {
-    return NUMBER_ERROR;
+    return INCORRECT_TYPE;
   }
-  return d;
-}
-
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * src) noexcept {
-  //
-  // Check for minus sign
-  //
-  bool negative = (*(src + 1) == '-');
-  src += uint8_t(negative) + 1;
-
-  //
-  // Parse the integer part.
-  //
-  uint64_t i = 0;
-  const uint8_t *p = src;
-  p += parse_digit(*p, i);
-  bool leading_zero = (i == 0);
-  while (parse_digit(*p, i)) { p++; }
-  // no integer digits, or 0123 (zero must be solo)
-  if ( p == src ) { return INCORRECT_TYPE; }
   if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }

   //
@@ -37944,9 +48341,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   if (simdjson_likely(*p == '.')) {
     p++;
     const uint8_t *start_decimal_digits = p;
-    if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
-    p++;
-    while (parse_digit(*p, i)) { p++; }
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
     exponent = -(p - start_decimal_digits);

     // Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -37985,9 +48381,9 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   //
   // Assemble (or slow-parse) the float
   //
-  double d;
+  float d;
   if (simdjson_likely(!overflow)) {
-    if (compute_float_64(exponent, i, negative, d)) { return d; }
+    if (compute_float_32(exponent, i, negative, d)) { return d; }
   }
   if (!parse_float_fallback(src - uint8_t(negative), &d)) {
     return NUMBER_ERROR;
@@ -38332,16 +48728,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
 }
 #endif

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
-                                uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
-  return _addcarry_u64(0, value1, value2,
-                       reinterpret_cast<unsigned __int64 *>(result));
-#else
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-#endif
-}

 } // unnamed namespace
 } // namespace westmere
@@ -38920,16 +49306,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
 }
 #endif

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
-                                uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
-  return _addcarry_u64(0, value1, value2,
-                       reinterpret_cast<unsigned __int64 *>(result));
-#else
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-#endif
-}

 } // unnamed namespace
 } // namespace westmere
@@ -40340,6 +50716,296 @@ simdjson_inline uint32_t find_next_document_index(dom_parser_implementation &par
   return 0;
 }

+/**
+ * Sentinel value returned to indicate a document started but didn't fit
+ * (CAPACITY error), as opposed to 0 which means no document content found
+ * (EMPTY).
+ */
+constexpr uint32_t DOCUMENT_TOO_LARGE = UINT32_MAX;
+
+/**
+ * For RFC 7464 JSON text sequences, filter RS from structural indexes and
+ * find batch boundaries.
+ *
+ * In JSON sequence mode, RS (0x1E) marks the start of each JSON text.
+ * RS bytes appear in structural_indexes as they are classified as scalars.
+ * This function:
+ * 1. Scans structural_indexes to find and count RS positions
+ * 2. Filters RS out of structural_indexes in-place
+ * 3. Determines batch boundaries based on RS positions
+ *
+ * @param parser The parser with structural_indexes and buf.
+ * @param len The length of the current batch buffer.
+ * @param is_final True if this is the final batch (no more data coming).
+ * @param next_batch_start Output: offset where the next batch should start.
+ * @param scan_len Offset past which no value start may be derived. Equals len
+ *        unless stage 1 dropped a trailing unclosed string, whose bytes it
+ *        never validated.
+ * @return The number of structural indexes to keep (after RS filtering),
+ *         0 if no document content found (EMPTY),
+ *         or DOCUMENT_TOO_LARGE if a document started but didn't fit (CAPACITY).
+ */
+simdjson_inline uint32_t find_next_document_index_json_sequence(
+    dom_parser_implementation &parser,
+    size_t len,
+    bool is_final,
+    uint32_t &next_batch_start,
+    size_t scan_len) {
+  // Default: next batch starts at end of buffer
+  next_batch_start = uint32_t(len);
+
+  if (parser.n_structural_indexes == 0) { return 0; }
+
+  // Phase 1: Scan structural_indexes to find RS positions and handle them.
+  // RS marks the start of a JSON text. For objects/arrays, the '{' or '[' after RS
+  // is already in structural_indexes (it's an operator). For scalars like numbers,
+  // the digit following RS is NOT in structural_indexes because the scanner sees
+  // RS as a scalar, making the digit a scalar continuation, not a start.
+  // We must: (1) remove RS from structural_indexes, and (2) for scalars, add the
+  // actual value start position.
+  // The EOF sentinel: len, or where a discarded unclosed string starts.
+  const uint32_t sentinel = parser.structural_indexes[parser.n_structural_indexes];
+  uint32_t write_idx = 0;
+  uint32_t last_rs_pos = 0;
+  uint32_t rs_count = 0;
+
+  for (uint32_t read_idx = 0; read_idx < parser.n_structural_indexes; read_idx++) {
+    const uint32_t pos = parser.structural_indexes[read_idx];
+    if (parser.buf[pos] == 0x1E) {
+      // This is an RS character - find the actual JSON value start.
+      last_rs_pos = pos;
+      rs_count++;
+      // Skip past this RS and any whitespace *and any additional RSes*
+      // to locate the real value. Consecutive RSes are degenerate
+      // "empty records" per RFC 7464; we collapse them here. They do
+      // not always appear as separate entries in structural_indexes
+      // because the scanner groups runs of adjacent non-whitespace
+      // scalar bytes (including RS) into a single scalar start.
+      uint32_t value_start = pos + 1;
+      while (value_start < scan_len) {
+        const uint8_t c = parser.buf[value_start];
+        if (c == ' ' || c == '\t' || c == '\n' || c == '\r') {
+          value_start++;
+        } else if (c == 0x1E) {
+          // Collapsed empty record. Still count it so rs_count reflects
+          // the true number of record markers and last_rs_pos tracks
+          // the final one.
+          last_rs_pos = value_start;
+          rs_count++;
+          value_start++;
+        } else {
+          break;
+        }
+      }
+      // If the scanner emitted additional structurals inside the
+      // whitespace+RS run we just walked over (i.e., isolated RSes
+      // separated by whitespace), skip past them so we do not
+      // double-count or double-emit.
+      while (read_idx + 1 < parser.n_structural_indexes &&
+             parser.structural_indexes[read_idx + 1] < value_start) {
+        read_idx++;
+      }
+      // Check if the value start is an operator (always present in
+      // scanner structural_indexes) or a scalar-like start (which may
+      // be missing from structural_indexes and must be added here).
+      // Note: '"' is NOT always in structural_indexes. The scanner
+      // classifies '"' as a scalar character and emits it as a
+      // structural only when it is a *scalar start* (preceded by
+      // whitespace or an operator). When '"' immediately follows an
+      // RS (which the scanner also classifies as scalar), it is
+      // treated as a scalar continuation and not emitted - so we
+      // must add it here just like any other scalar value.
+      if (value_start < scan_len) {
+        const uint8_t c = parser.buf[value_start];
+        const bool is_operator =
+            (c == '{' || c == '}' || c == '[' || c == ']' ||
+             c == ':' || c == ',');
+        // If the next scanner structural is exactly at value_start,
+        // the scanner already emitted it (it followed whitespace) and
+        // we must not add a duplicate - a subsequent iteration will
+        // copy it into write_idx.
+        const bool already_emitted =
+            (read_idx + 1 < parser.n_structural_indexes &&
+             parser.structural_indexes[read_idx + 1] == value_start);
+        if (!is_operator && !already_emitted) {
+          // Scalar value (number/true/false/null/string) - add its
+          // position since scanner missed it.
+          parser.structural_indexes[write_idx++] = value_start;
+        }
+      }
+    } else {
+      // Not RS, copy to output
+      parser.structural_indexes[write_idx++] = pos;
+    }
+  }
+
+  // Update structural index count. Compaction left a stale index in the slot
+  // past the end: restore the EOF sentinel that stage 1 had planted there, which
+  // document_stream::truncated_bytes() reads after a final batch.
+  parser.n_structural_indexes = write_idx;
+  parser.structural_indexes[write_idx] = sentinel;
+
+  if (parser.n_structural_indexes == 0) {
+    // Only RS markers here: the last one opens a record continuing past the
+    // window, so that is where the next batch resumes.
+    if (!is_final && rs_count > 0) { next_batch_start = last_rs_pos; }
+    return 0;
+  }
+  if (rs_count == 0) {
+    // No RS found; for final batch, try generic boundary detection
+    return is_final ? find_next_document_index(parser) : 0;
+  }
+
+  // Phase 2: Determine batch boundaries based on RS positions
+
+  if (is_final) {
+    // Final batch: all documents are complete (last one ends at EOF).
+    // In json_sequence mode, RS markers define document boundaries, so all
+    // remaining structurals form complete documents. Return them all directly.
+    // (Calling find_next_document_index() would fail for scalar documents.)
+    return parser.n_structural_indexes;
+  }
+
+  // Partial batch: need to find complete documents only.
+  // A document starting at an RS is complete if there is another RS after it.
+  next_batch_start = last_rs_pos;
+
+  if (rs_count < 2) {
+    // Only one RS, so we have at most one document that may be incomplete.
+    // We cannot confirm it is complete without another RS.
+    // Return DOCUMENT_TOO_LARGE if content was found (write_idx > 0), 0 if only separators.
+    return (parser.n_structural_indexes > 0) ? DOCUMENT_TOO_LARGE : 0;
+  }
+
+  // We have at least 2 RS markers. The last complete document ends before last_rs_pos.
+
+  // Find the structural index cutoff: keep only structurals < last_rs_pos.
+  // Since we already filtered RS, all remaining structurals are valid.
+  // We iterate backward to find the last structural before last_rs_pos.
+  uint32_t keep_count = 0;
+  for (uint32_t i = parser.n_structural_indexes; i > 0; i--) {
+    if (parser.structural_indexes[i - 1] < last_rs_pos) {
+      keep_count = i;
+      break;
+    }
+  }
+
+  // No structurals before the last RS - no complete documents
+  if (keep_count == 0) { return 0; }
+
+  // All documents before the last RS are complete by definition (the next RS
+  // confirms their end). No need to call find_next_document_index() which
+  // would fail for scalar documents like `1` or `"hello"`.
+  return keep_count;
+}
+
+/**
+ * Filter comma-delimited documents by removing root-level commas from
+ * structural indexes.
+ *
+ * For comma-delimited format like `{...},{...},{...}`, we need to remove
+ * the commas that separate documents (depth 0) while preserving commas
+ * inside arrays and objects (depth > 0).
+ *
+ * After filtering, the structural indexes look like whitespace-delimited
+ * documents, so find_next_document_index() works unchanged.
+ *
+ * @param parser The parser with structural_indexes and buf.
+ * @param len The length of the current batch buffer.
+ * @param is_final True if this is the final batch (no more data coming).
+ * @param next_batch_start Output: offset where the next batch should start.
+ * @return The number of structural indexes to keep,
+ *         0 if no document content found (EMPTY),
+ *         or DOCUMENT_TOO_LARGE if a document started but didn't fit (CAPACITY).
+ */
+simdjson_inline uint32_t filter_comma_delimited(
+    dom_parser_implementation &parser,
+    size_t len,
+    bool is_final,
+    uint32_t &next_batch_start) {
+  // Default: next batch starts at end of buffer
+  next_batch_start = uint32_t(len);
+
+  if (parser.n_structural_indexes == 0) { return 0; }
+
+  // Track depth to identify root-level commas (depth 0)
+  int depth = 0;
+  // The EOF sentinel: len, or where a discarded unclosed string starts.
+  const uint32_t sentinel = parser.structural_indexes[parser.n_structural_indexes];
+  uint32_t write_idx = 0;
+  uint32_t last_root_comma_pos = 0;
+  uint32_t root_comma_count = 0;
+
+  for (uint32_t i = 0; i < parser.n_structural_indexes; i++) {
+    uint32_t idx = parser.structural_indexes[i];
+    uint8_t c = parser.buf[idx];
+
+    switch (c) {
+      case '{': case '[':
+        depth++;
+        parser.structural_indexes[write_idx++] = idx;
+        break;
+      case '}': case ']':
+        depth--;
+        parser.structural_indexes[write_idx++] = idx;
+        break;
+      case ',':
+        if (depth == 0) {
+          // Root-level comma = document boundary, skip it
+          last_root_comma_pos = idx;
+          root_comma_count++;
+          continue;
+        }
+        parser.structural_indexes[write_idx++] = idx;
+        break;
+      default:
+        // Colons, scalars, etc.
+        parser.structural_indexes[write_idx++] = idx;
+        break;
+    }
+  }
+
+  // Update structural index count. Compaction left a stale index in the slot
+  // past the end: restore the EOF sentinel that stage 1 had planted there, which
+  // document_stream::truncated_bytes() reads after a final batch.
+  parser.n_structural_indexes = write_idx;
+  parser.structural_indexes[write_idx] = sentinel;
+
+  if (parser.n_structural_indexes == 0) { return 0; }
+
+  if (is_final) {
+    // Final batch: use standard boundary detection on filtered indexes
+    return find_next_document_index(parser);
+  }
+
+  // Partial batch: need to find complete documents only.
+  // A document ending with a root comma is complete.
+  if (root_comma_count == 0) {
+    // No root commas found; we cannot confirm any document is complete.
+    // The whole batch might be one incomplete document.
+    // Return DOCUMENT_TOO_LARGE if content was found (write_idx > 0), 0 if only commas.
+    return (parser.n_structural_indexes > 0) ? DOCUMENT_TOO_LARGE : 0;
+  }
+
+  // We have at least one root comma. Documents before the last comma are complete.
+  next_batch_start = last_root_comma_pos + 1;
+
+  // Find the structural index cutoff: keep only structurals < last_root_comma_pos
+  uint32_t keep_count = 0;
+  for (uint32_t i = parser.n_structural_indexes; i > 0; i--) {
+    if (parser.structural_indexes[i - 1] < last_root_comma_pos) {
+      keep_count = i;
+      break;
+    }
+  }
+
+  if (keep_count == 0) { return 0; }
+
+  // Use standard boundary detection on the complete portion
+  parser.n_structural_indexes = keep_count;
+  return find_next_document_index(parser);
+}
+
 } // namespace stage1
 } // unnamed namespace
 } // namespace westmere
@@ -40773,7 +51439,6 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
         return EMPTY;
       }
     }
-
     parser.n_structural_indexes = new_structural_indexes;
   } else if (partial == stage1_mode::streaming_final) {
     if(have_unclosed_string) { parser.n_structural_indexes--; }
@@ -40801,6 +51466,75 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
         // the trailing garbage.
         return EMPTY;
     }
+  } else if (partial == stage1_mode::json_sequence_partial) {
+    // RFC 7464: use RS positions for batch boundaries
+    // A discarded unclosed string also caps how far the filter may scan.
+    size_t scan_len = len;
+    if(have_unclosed_string) {
+      parser.n_structural_indexes--;
+      if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return CAPACITY; }
+      scan_len = parser.structural_indexes[parser.n_structural_indexes];
+    }
+    uint32_t next_batch_start = uint32_t(len);
+    auto new_structural_indexes = find_next_document_index_json_sequence(parser, len, false, next_batch_start, scan_len);
+    if (new_structural_indexes == DOCUMENT_TOO_LARGE) {
+      return CAPACITY;
+    }
+    if (new_structural_indexes == 0) {
+      // An EMPTY batch must still advance next_batch_start, or document_stream
+      // re-parses the same bytes forever. CAPACITY when it cannot.
+      if (next_batch_start == 0) { return CAPACITY; }
+      parser.n_structural_indexes = 0;
+      parser.structural_indexes[0] = next_batch_start;
+      return EMPTY;
+    }
+    parser.n_structural_indexes = new_structural_indexes;
+    parser.structural_indexes[parser.n_structural_indexes] = next_batch_start;
+  } else if (partial == stage1_mode::json_sequence_final) {
+    // RFC 7464: final batch, last document extends to EOF
+    // As above: a discarded unclosed string caps how far the filter may scan.
+    size_t scan_len = len;
+    if(have_unclosed_string) {
+      parser.n_structural_indexes--;
+      scan_len = parser.structural_indexes[parser.n_structural_indexes];
+    }
+    uint32_t next_batch_start = uint32_t(len);
+    parser.n_structural_indexes = find_next_document_index_json_sequence(parser, len, true, next_batch_start, scan_len);
+    // The filter compacted structural_indexes in place and restored the EOF
+    // sentinel past the compacted end, so the copy below is either the start
+    // of a truncated document or len, as in streaming_final.
+    parser.structural_indexes[parser.n_structural_indexes + 1] = parser.structural_indexes[parser.n_structural_indexes];
+    parser.structural_indexes[parser.n_structural_indexes] = uint32_t(len);
+    if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return EMPTY; }
+  } else if (partial == stage1_mode::comma_delimited_partial) {
+    // Comma-delimited: filter root-level commas, use comma positions for batch boundaries
+    if(have_unclosed_string) {
+      parser.n_structural_indexes--;
+      if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return CAPACITY; }
+    }
+    uint32_t next_batch_start = uint32_t(len);
+    auto new_structural_indexes = filter_comma_delimited(parser, len, false, next_batch_start);
+    if (new_structural_indexes == DOCUMENT_TOO_LARGE) {
+      return CAPACITY;
+    }
+    if (new_structural_indexes == 0) {
+      // An EMPTY batch must still advance next_batch_start, or document_stream
+      // re-parses the same bytes forever. CAPACITY when it cannot.
+      if (next_batch_start == 0) { return CAPACITY; }
+      parser.n_structural_indexes = 0;
+      parser.structural_indexes[0] = next_batch_start;
+      return EMPTY;
+    }
+    parser.n_structural_indexes = new_structural_indexes;
+    parser.structural_indexes[parser.n_structural_indexes] = next_batch_start;
+  } else if (partial == stage1_mode::comma_delimited_final) {
+    // Comma-delimited: final batch, last document extends to EOF
+    if(have_unclosed_string) { parser.n_structural_indexes--; }
+    uint32_t next_batch_start = uint32_t(len);
+    parser.n_structural_indexes = filter_comma_delimited(parser, len, true, next_batch_start);
+    parser.structural_indexes[parser.n_structural_indexes + 1] = parser.structural_indexes[parser.n_structural_indexes];
+    parser.structural_indexes[parser.n_structural_indexes] = uint32_t(len);
+    if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return EMPTY; }
   }
   checker.check_eof();
   return checker.errors();
@@ -40883,7 +51617,6 @@ namespace {
 namespace stage2 {

 class json_iterator;
-class structural_iterator;
 struct tape_builder;
 struct tape_writer;

@@ -41180,7 +51913,7 @@ public:
    *
    * - increment_count(iter) - each time a value is found in an array or object.
    */
-  template<bool STREAMING, typename V>
+  template<bool STREAMING, bool UNPADDED, typename V>
   simdjson_warn_unused simdjson_inline error_code walk_document(V &visitor) noexcept;

   /**
@@ -41197,6 +51930,7 @@ public:
    *
    * They may include invalid JSON as well (such as `1.2.3` or `ture`).
    */
+  template<bool UNPADDED>
   simdjson_inline const uint8_t *peek() const noexcept;
   /**
    * Advance to the next token.
@@ -41205,6 +51939,7 @@ public:
    *
    * They may include invalid JSON as well (such as `1.2.3` or `ture`).
    */
+  template<bool UNPADDED>
   simdjson_inline const uint8_t *advance() noexcept;
   /**
    * Get the remaining length of the document, from the start of the current token.
@@ -41253,7 +51988,7 @@ public:
   simdjson_warn_unused simdjson_inline error_code visit_primitive(V &visitor, const uint8_t *value) noexcept;
 };

-template<bool STREAMING, typename V>
+template<bool STREAMING, bool UNPADDED, typename V>
 simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &visitor) noexcept {
   logger::log_start();

@@ -41268,7 +52003,7 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &
   // Read first value
   //
   {
-    auto value = advance();
+    auto value = advance<UNPADDED>();

     // Make sure the outer object or array is closed before continuing; otherwise, there are ways we
     // could get into memory corruption. See https://github.com/simdjson/simdjson/issues/906
@@ -41280,8 +52015,8 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &
     }

     switch (*value) {
-      case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
-      case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+      case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+      case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
       default: SIMDJSON_TRY( visitor.visit_root_primitive(*this, value) ); break;
     }
   }
@@ -41298,29 +52033,29 @@ object_begin:
   SIMDJSON_TRY( visitor.visit_object_start(*this) );

   {
-    auto key = advance();
+    auto key = advance<UNPADDED>();
     if (*key != '"') { log_error("Object does not start with a key"); return TAPE_ERROR; }
     SIMDJSON_TRY( visitor.increment_count(*this) );
     SIMDJSON_TRY( visitor.visit_key(*this, key) );
   }

 object_field:
-  if (simdjson_unlikely( *advance() != ':' )) { log_error("Missing colon after key in object"); return TAPE_ERROR; }
+  if (simdjson_unlikely( *advance<UNPADDED>() != ':' )) { log_error("Missing colon after key in object"); return TAPE_ERROR; }
   {
-    auto value = advance();
+    auto value = advance<UNPADDED>();
     switch (*value) {
-      case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
-      case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+      case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+      case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
       default: SIMDJSON_TRY( visitor.visit_primitive(*this, value) ); break;
     }
   }

 object_continue:
-  switch (*advance()) {
+  switch (*advance<UNPADDED>()) {
     case ',':
       SIMDJSON_TRY( visitor.increment_count(*this) );
       {
-        auto key = advance();
+        auto key = advance<UNPADDED>();
         if (simdjson_unlikely( *key != '"' )) { log_error("Key string missing at beginning of field in object"); return TAPE_ERROR; }
         SIMDJSON_TRY( visitor.visit_key(*this, key) );
       }
@@ -41348,16 +52083,16 @@ array_begin:

 array_value:
   {
-    auto value = advance();
+    auto value = advance<UNPADDED>();
     switch (*value) {
-      case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
-      case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+      case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+      case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
       default: SIMDJSON_TRY( visitor.visit_primitive(*this, value) ); break;
     }
   }

 array_continue:
-  switch (*advance()) {
+  switch (*advance<UNPADDED>()) {
     case ',': SIMDJSON_TRY( visitor.increment_count(*this) ); goto array_value;
     case ']': log_end_value("array"); SIMDJSON_TRY( visitor.visit_array_end(*this) ); goto scope_end;
     default: log_error("Missing comma between array values"); return TAPE_ERROR;
@@ -41385,11 +52120,29 @@ simdjson_inline json_iterator::json_iterator(dom_parser_implementation &_dom_par
     dom_parser{_dom_parser} {
 }

+// Stage 1 leaves a sentinel index of value `len`; reading buf[len] requires
+// padding. For UNPADDED, return a dummy so walk_document can fail cleanly (#2815).
+template<bool UNPADDED>
 simdjson_inline const uint8_t *json_iterator::peek() const noexcept {
-  return &buf[*(next_structural)];
+  const uint32_t idx = *(next_structural);
+  SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+    if (simdjson_unlikely(idx >= dom_parser.len)) {
+      static constexpr uint8_t unpadded_eof_sentinel = 0;
+      return &unpadded_eof_sentinel;
+    }
+  }
+  return &buf[idx];
 }
+template<bool UNPADDED>
 simdjson_inline const uint8_t *json_iterator::advance() noexcept {
-  return &buf[*(next_structural++)];
+  const uint32_t idx = *(next_structural++);
+  SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+    if (simdjson_unlikely(idx >= dom_parser.len)) {
+      static constexpr uint8_t unpadded_eof_sentinel = 0;
+      return &unpadded_eof_sentinel;
+    }
+  }
+  return &buf[idx];
 }
 simdjson_inline size_t json_iterator::remaining_len() const noexcept {
   return dom_parser.len - *(next_structural-1);
@@ -41429,7 +52182,20 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::visit_root_primit
     case '"': return visitor.visit_root_string(*this, value);
     case 't': return visitor.visit_root_true_atom(*this, value);
     case 'f': return visitor.visit_root_false_atom(*this, value);
+#if SIMDJSON_ENABLE_NAN_INF
+    case 'n': {
+      auto err = visitor.visit_root_null_atom(*this, value);
+      if (err == SUCCESS) { return err; }
+      // propagate the error value returned by a bad 'null' atom if parsing 'nan' fails
+      return visitor.visit_root_nan_atom(*this, value, err);
+    }
+    // 'N' isn't a canonically recognized atom, so we return a TAPE_ERROR if failure occurs
+    case 'N': return visitor.visit_root_nan_atom(*this, value, TAPE_ERROR);
+    case 'i':
+    case 'I': return visitor.visit_root_inf_atom(*this, value);
+#else
     case 'n': return visitor.visit_root_null_atom(*this, value);
+#endif
     case '-':
     case '0': case '1': case '2': case '3': case '4':
     case '5': case '6': case '7': case '8': case '9':
@@ -41451,7 +52217,20 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::visit_primitive(V
   switch (*value) {
     case 't': return visitor.visit_true_atom(*this, value);
     case 'f': return visitor.visit_false_atom(*this, value);
+#if SIMDJSON_ENABLE_NAN_INF
+    case 'n': {
+      auto err = visitor.visit_null_atom(*this, value);
+      if (err == SUCCESS) { return err; }
+      // propagate the error value returned by a bad 'null' atom if parsing 'nan' fails
+      return visitor.visit_nan_atom(*this, value, err);
+    }
+    // 'N' isn't a canonically recognized atom, so we return a TAPE_ERROR if failure occurs
+    case 'N': return visitor.visit_nan_atom(*this, value, TAPE_ERROR);
+    case 'i':
+    case 'I': return visitor.visit_inf_atom(*this, value);
+#else
     case 'n': return visitor.visit_null_atom(*this, value);
+#endif
     default:
       log_error("Non-value found when value was expected!");
       return TAPE_ERROR;
@@ -41637,9 +52416,13 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_string(const uint8_t *src, u
            within the unicode codepoint handling code. */
         src += bs_dist;
         dst += bs_dist;
-        if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
-          return nullptr;
-        }
+        // Decode adjacent Unicode escapes without returning to the
+        // quote-and-backslash scanner between code points.
+        do {
+          if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
+            return nullptr;
+          }
+        } while (src[0] == '\\' && src[1] == 'u');
       } else {
         /* simple 1:1 conversion. Will eat bs_dist+2 characters in input and
          * write bs_dist+1 characters to output
@@ -41662,6 +52445,77 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_string(const uint8_t *src, u
   }
 }

+/**
+ * Bounds-safe variant of parse_string for input buffers that are NOT padded to
+ * len + SIMDJSON_PADDING bytes. `buf_end` is one past the last readable input
+ * byte (buf + len). It behaves exactly like parse_string while we are at least
+ * SIMDJSON_PADDING bytes away from buf_end (so every speculative SIMD read stays
+ * in bounds); once within the final SIMDJSON_PADDING bytes it copies the few
+ * remaining bytes into a space-padded scratch buffer and finishes with the
+ * regular parse_string. This keeps the delicate escape/Unicode handling in one
+ * place (the proven parse_string) rather than duplicating it.
+ *
+ * Correctness relies on stage 1 having validated the string, i.e. there is an
+ * unescaped closing quote within [src, buf_end); that quote is therefore inside
+ * the copied scratch, so parse_string finds it without running off the scratch.
+ */
+simdjson_warn_unused simdjson_inline uint8_t *parse_string_safe(const uint8_t *src, uint8_t *dst, bool allow_replacement, const uint8_t *buf_end) {
+  // Far from the end: identical to parse_string's loop. The guard uses
+  // SIMDJSON_PADDING (>= BYTES_PROCESSED) so copy_and_find never reads past
+  // buf_end; escape/Unicode look-aheads read within the string (before the
+  // closing quote, which is < buf_end), so they are in bounds here too.
+  // We add margin (+12) for handle_unicode_codepoint's worst-case lookahead:
+  // after seeing a high surrogate, it does hex_to_u32_nocheck on the immediate
+  // following bytes (+6 from the '\'), then (if it sees \u) another
+  // hex_to_u32_nocheck at +8..+11 relative to the backslash that started the
+  // escape. With bs_dist up to BYTES_PROCESSED-1 this reaches +11 from the
+  // chunk start. The +12 margin ensures that even on kernels where
+  // BYTES_PROCESSED == SIMDJSON_PADDING (e.g. icelake) the 4-byte read stays
+  // in-bounds. The scratch fallback (3*PAD) is already safe.
+  while (src + SIMDJSON_PADDING + 12 <= buf_end) {
+    auto b = backslash_and_quote{};
+    auto bs_quote = b.copy_and_find(src, dst);
+    if (bs_quote.has_quote_first()) {
+      return dst + bs_quote.quote_index();
+    }
+    if (bs_quote.has_backslash()) {
+      auto bs_dist = bs_quote.backslash_index();
+      uint8_t escape_char = src[bs_dist + 1];
+      if (escape_char == 'u') {
+        src += bs_dist;
+        dst += bs_dist;
+        if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
+          return nullptr;
+        }
+      } else {
+        uint8_t escape_result = escape_map[escape_char];
+        if (escape_result == 0u) {
+          return nullptr;
+        }
+        dst[bs_dist] = escape_result;
+        src += bs_dist + 2;
+        dst += bs_dist + 1;
+      }
+    } else {
+      src += backslash_and_quote::BYTES_PROCESSED;
+      dst += backslash_and_quote::BYTES_PROCESSED;
+    }
+  }
+  // Within the final SIMDJSON_PADDING bytes: copy what remains into a
+  // space-padded scratch (spaces are neither quote nor backslash, so they do not
+  // disturb matching) and let the regular parser finish from there. The closing
+  // quote is within `remaining` (< SIMDJSON_PADDING), so parse_string finds it in
+  // the chunk starting at some offset <= remaining and reads at most
+  // BYTES_PROCESSED (<= SIMDJSON_PADDING) further -- i.e. under 2*SIMDJSON_PADDING.
+  // We size at 3x for a comfortable margin (the unicode look-ahead reads a few
+  // extra bytes past an escape).
+  uint8_t scratch[SIMDJSON_PADDING * 3];
+  const size_t remaining = size_t(buf_end - src); // < SIMDJSON_PADDING
+  std::memset(scratch, ' ', sizeof(scratch));
+  std::memcpy(scratch, src, remaining);
+  return parse_string(scratch, dst, allow_replacement);
+}
+
 simdjson_warn_unused simdjson_inline uint8_t *parse_wobbly_string(const uint8_t *src, uint8_t *dst) {
   // It is not ideal that this function is nearly identical to parse_string.
   while (1) {
@@ -41716,73 +52570,6 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_wobbly_string(const uint8_t

 #endif // SIMDJSON_SRC_GENERIC_STAGE2_STRINGPARSING_H
 /* end file generic/stage2/stringparsing.h for westmere */
-/* including generic/stage2/structural_iterator.h for westmere: #include <generic/stage2/structural_iterator.h> */
-/* begin file generic/stage2/structural_iterator.h for westmere */
-#ifndef SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H
-
-/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
-/* amalgamation skipped (editor-only): #define SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H */
-/* amalgamation skipped (editor-only): #include <generic/stage2/base.h> */
-/* amalgamation skipped (editor-only): #include <simdjson/generic/dom_parser_implementation.h> */
-/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
-
-namespace simdjson {
-namespace westmere {
-namespace {
-namespace stage2 {
-
-class structural_iterator {
-public:
-  const uint8_t* const buf;
-  uint32_t *next_structural;
-  dom_parser_implementation &dom_parser;
-
-  // Start a structural
-  simdjson_inline structural_iterator(dom_parser_implementation &_dom_parser, size_t start_structural_index)
-    : buf{_dom_parser.buf},
-      next_structural{&_dom_parser.structural_indexes[start_structural_index]},
-      dom_parser{_dom_parser} {
-  }
-  // Get the buffer position of the current structural character
-  simdjson_inline const uint8_t* current() {
-    return &buf[*(next_structural-1)];
-  }
-  // Get the current structural character
-  simdjson_inline char current_char() {
-    return buf[*(next_structural-1)];
-  }
-  // Get the next structural character without advancing
-  simdjson_inline char peek_next_char() {
-    return buf[*next_structural];
-  }
-  simdjson_inline const uint8_t* peek() {
-    return &buf[*next_structural];
-  }
-  simdjson_inline const uint8_t* advance() {
-    return &buf[*(next_structural++)];
-  }
-  simdjson_inline char advance_char() {
-    return buf[*(next_structural++)];
-  }
-  simdjson_inline size_t remaining_len() {
-    return dom_parser.len - *(next_structural-1);
-  }
-
-  simdjson_inline bool at_end() {
-    return next_structural == &dom_parser.structural_indexes[dom_parser.n_structural_indexes];
-  }
-  simdjson_inline bool at_beginning() {
-    return next_structural == dom_parser.structural_indexes.get();
-  }
-};
-
-} // namespace stage2
-} // unnamed namespace
-} // namespace westmere
-} // namespace simdjson
-
-#endif // SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H
-/* end file generic/stage2/structural_iterator.h for westmere */
 /* including generic/stage2/tape_builder.h for westmere: #include <generic/stage2/tape_builder.h> */
 /* begin file generic/stage2/tape_builder.h for westmere */
 #ifndef SIMDJSON_SRC_GENERIC_STAGE2_TAPE_BUILDER_H
@@ -41805,12 +52592,8 @@ namespace westmere {
 namespace {
 namespace stage2 {

-struct tape_builder {
-  template<bool STREAMING>
-  simdjson_warn_unused static simdjson_inline error_code parse_document(
-    dom_parser_implementation &dom_parser,
-    dom::document &doc) noexcept;
-
+template <bool UNPADDED>
+struct tape_builder_impl {
   /** Called when a non-empty document starts. */
   simdjson_warn_unused simdjson_inline error_code visit_document_start(json_iterator &iter) noexcept;
   /** Called when a non-empty document ends without error. */
@@ -41863,88 +52646,130 @@ struct tape_builder {
   simdjson_warn_unused simdjson_inline error_code visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept;
   simdjson_warn_unused simdjson_inline error_code visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept;

+#if SIMDJSON_ENABLE_NAN_INF
+  simdjson_warn_unused simdjson_inline error_code visit_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept;
+  simdjson_warn_unused simdjson_inline error_code visit_root_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept;
+  // Attempts to parse 'inf' or 'infinity' (case insensitive). Because neither are canonical atoms,
+  // this returns a tape error on failure.
+  simdjson_warn_unused simdjson_inline error_code visit_inf_atom(json_iterator &iter, const uint8_t *value) noexcept;
+  simdjson_warn_unused simdjson_inline error_code visit_root_inf_atom(json_iterator &iter, const uint8_t *value) noexcept;
+#endif
+
   /** Called each time a new field or element in an array or object is found. */
   simdjson_warn_unused simdjson_inline error_code increment_count(json_iterator &iter) noexcept;

   /** Next location to write to tape */
   tape_writer tape;
+public:
+  simdjson_inline tape_builder_impl(dom::document &doc) noexcept;
 private:
   /** Next write location in the string buf for stage 2 parsing */
   uint8_t *current_string_buf_loc;

-  simdjson_inline tape_builder(dom::document &doc) noexcept;
-
   simdjson_inline uint32_t next_tape_index(json_iterator &iter) const noexcept;
   simdjson_inline void start_container(json_iterator &iter) noexcept;
   simdjson_warn_unused simdjson_inline error_code end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept;
   simdjson_warn_unused simdjson_inline error_code empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept;
   simdjson_inline uint8_t *on_start_string(json_iterator &iter) noexcept;
   simdjson_inline void on_end_string(uint8_t *dst) noexcept;
-}; // struct tape_builder
+}; // struct tape_builder_impl

-template<bool STREAMING>
-simdjson_warn_unused simdjson_inline error_code tape_builder::parse_document(
-    dom_parser_implementation &dom_parser,
-    dom::document &doc) noexcept {
-  dom_parser.doc = &doc;
-  json_iterator iter(dom_parser, STREAMING ? dom_parser.next_structural_index : 0);
-  tape_builder builder(doc);
-  return iter.walk_document<STREAMING>(builder);
-}
+// Thin, non-templated entry so each architecture's stage2() keeps calling
+// tape_builder::parse_document<STREAMING> unchanged. It chooses the bounds-safe
+// (unpadded) or the regular (padded) tape_builder_impl ONCE per document, so the
+// choice is a compile-time constant inside the walk: the padded path carries no
+// extra branch or load (see tape_builder_impl::visit_string).
+struct tape_builder {
+  template<bool STREAMING>
+  simdjson_warn_unused static simdjson_inline error_code parse_document(
+      dom_parser_implementation &dom_parser, dom::document &doc) noexcept {
+    dom_parser.doc = &doc;
+    json_iterator iter(dom_parser, STREAMING ? dom_parser.next_structural_index : 0);
+    if (dom_parser._unpadded) {
+      tape_builder_impl<true> builder(doc);
+      return iter.walk_document<STREAMING, true>(builder);
+    } else {
+      tape_builder_impl<false> builder(doc);
+      return iter.walk_document<STREAMING, false>(builder);
+    }
+  }
+};

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_primitive(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_primitive(json_iterator &iter, const uint8_t *value) noexcept {
   return iter.visit_root_primitive(*this, value);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_primitive(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_primitive(json_iterator &iter, const uint8_t *value) noexcept {
   return iter.visit_primitive(*this, value);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_empty_object(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_empty_object(json_iterator &iter) noexcept {
   return empty_container(iter, internal::tape_type::START_OBJECT, internal::tape_type::END_OBJECT);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_empty_array(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_empty_array(json_iterator &iter) noexcept {
   return empty_container(iter, internal::tape_type::START_ARRAY, internal::tape_type::END_ARRAY);
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_document_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_document_start(json_iterator &iter) noexcept {
   start_container(iter);
   return SUCCESS;
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_object_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_object_start(json_iterator &iter) noexcept {
   start_container(iter);
   return SUCCESS;
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_array_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_array_start(json_iterator &iter) noexcept {
   start_container(iter);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_object_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_object_end(json_iterator &iter) noexcept {
   return end_container(iter, internal::tape_type::START_OBJECT, internal::tape_type::END_OBJECT);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_array_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_array_end(json_iterator &iter) noexcept {
   return end_container(iter, internal::tape_type::START_ARRAY, internal::tape_type::END_ARRAY);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_document_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_document_end(json_iterator &iter) noexcept {
   constexpr uint32_t start_tape_index = 0;
   tape.append(start_tape_index, internal::tape_type::ROOT);
   tape_writer::write(iter.dom_parser.doc->tape[start_tape_index], next_tape_index(iter), internal::tape_type::ROOT);
   return SUCCESS;
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_key(json_iterator &iter, const uint8_t *key) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_key(json_iterator &iter, const uint8_t *key) noexcept {
   return visit_string(iter, key, true);
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::increment_count(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::increment_count(json_iterator &iter) noexcept {
   iter.dom_parser.open_containers[iter.depth].count++; // we have a key value pair in the object at parser.dom_parser.depth - 1
   return SUCCESS;
 }

-simdjson_inline tape_builder::tape_builder(dom::document &doc) noexcept : tape{doc.tape.get()}, current_string_buf_loc{doc.string_buf.get()} {}
+template <bool UNPADDED>
+simdjson_inline tape_builder_impl<UNPADDED>::tape_builder_impl(dom::document &doc) noexcept : tape{doc.tape.get()}, current_string_buf_loc{doc.string_buf.get()} {}

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_string(json_iterator &iter, const uint8_t *value, bool key) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_string(json_iterator &iter, const uint8_t *value, bool key) noexcept {
   iter.log_value(key ? "key" : "string");
   uint8_t *dst = on_start_string(iter);
-  dst = stringparsing::parse_string(value+1, dst, false); // We do not allow replacement when the escape characters are invalid.
+  // We do not allow replacement when the escape characters are invalid.
+  // UNPADDED is a compile-time constant chosen once per document by
+  // tape_builder::parse_document, so the padded build instantiates only the
+  // plain parse_string call below -- no runtime branch and no flag load.
+  SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+    dst = stringparsing::parse_string_safe(value+1, dst, false, iter.buf + iter.dom_parser.len);
+  } else {
+    dst = stringparsing::parse_string(value+1, dst, false);
+  }
   if (dst == nullptr) {
     iter.log_error("Invalid escape in string");
     return STRING_ERROR;
@@ -41953,27 +52778,48 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_string(json_
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_string(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_string(json_iterator &iter, const uint8_t *value) noexcept {
   return visit_string(iter, value);
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_number(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_number(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("number");
-  error_code err = numberparsing::parse_number(value, tape);
+  const uint8_t *num = value;
+  std::unique_ptr<uint8_t[]> copy{}; // keeps a padded copy of the tail alive when used
+  SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+    // numberparsing reads ahead in 8-byte blocks for floats
+    // (is_made_of_eight_digits_fast reads up to 7 bytes past the digits), so a
+    // number whose digits reach the final bytes of an unpadded buffer would read
+    // past it. *(next_structural) is the offset of the token following this
+    // number, hence an upper bound on where the digits end; when that is within
+    // SIMDJSON_PADDING of the end we parse from a space-padded copy of the tail
+    // (mirroring visit_root_number). This fires only for numbers near the end.
+    if (simdjson_unlikely(*(iter.next_structural) + SIMDJSON_PADDING > iter.dom_parser.len)) {
+      const size_t rl = iter.remaining_len(); // bytes from `value` to the end of the document
+      copy.reset(new (std::nothrow) uint8_t[rl + SIMDJSON_PADDING]);
+      if (copy.get() == nullptr) { return MEMALLOC; }
+      std::memcpy(copy.get(), value, rl);
+      std::memset(copy.get() + rl, ' ', SIMDJSON_PADDING);
+      num = copy.get();
+    }
+  }
+  error_code err = numberparsing::parse_number(num, tape);
   if (simdjson_unlikely(err == BIGINT_ERROR &&
       iter.dom_parser._number_as_string)) {
     // Write big integer to string buffer using the same format as strings.
     // Scan digits the same way parse_number does (skip optional '-', then digits).
-    const uint8_t *p = value;
+    const uint8_t *p = num;
     if (*p == '-') p++;
     while (numberparsing::is_digit(*p)) p++;
     // The digit run must be terminated by a structural or whitespace character; otherwise the
     // token is malformed (e.g. "123456789123456789123x").
     if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
-    size_t len = size_t(p - value);
+    size_t len = size_t(p - num);
     tape.append(current_string_buf_loc - iter.dom_parser.doc->string_buf.get(), internal::tape_type::BIGINT);
     uint8_t *dst = current_string_buf_loc + sizeof(uint32_t);
-    memcpy(dst, value, len);
+    memcpy(dst, num, len);
     dst += len;
     on_end_string(dst);
     return SUCCESS;
@@ -41981,7 +52827,8 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_number(json_
   return err;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_number(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_number(json_iterator &iter, const uint8_t *value) noexcept {
   //
   // We need to make a copy to make sure that the string is space terminated.
   // This is not about padding the input, which should already padded up
@@ -41995,76 +52842,142 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_number(
   // practice unless you are in the strange scenario where you have many JSON
   // documents made of single atoms.
   //
-  std::unique_ptr<uint8_t[]>copy(new (std::nothrow) uint8_t[iter.remaining_len() + SIMDJSON_PADDING]);
+  // In a stream, the input goes on with other documents: copy up to the next
+  // structural only, not to the end of the batch.
+  const size_t len = (std::min)(iter.remaining_len(), size_t(*iter.next_structural) - size_t(*(iter.next_structural - 1)));
+  std::unique_ptr<uint8_t[]>copy(new (std::nothrow) uint8_t[len + SIMDJSON_PADDING]);
   if (copy.get() == nullptr) { return MEMALLOC; }
-  std::memcpy(copy.get(), value, iter.remaining_len());
-  std::memset(copy.get() + iter.remaining_len(), ' ', SIMDJSON_PADDING);
+  std::memcpy(copy.get(), value, len);
+  std::memset(copy.get() + len, ' ', SIMDJSON_PADDING);
   error_code error = visit_number(iter, copy.get());
   return error;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("true");
-  if (!atomparsing::is_valid_true_atom(value)) { return T_ATOM_ERROR; }
+  // The non-length-aware validator reads a fixed 5 bytes; a malformed/truncated
+  // token at the very end of an unpadded buffer would over-read. Use the
+  // length-aware form there (the root variant already does this).
+  const bool ok = UNPADDED ? atomparsing::is_valid_true_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_true_atom(value);
+  if (!ok) { return T_ATOM_ERROR; }
   tape.append(0, internal::tape_type::TRUE_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("true");
   if (!atomparsing::is_valid_true_atom(value, iter.remaining_len())) { return T_ATOM_ERROR; }
   tape.append(0, internal::tape_type::TRUE_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("false");
-  if (!atomparsing::is_valid_false_atom(value)) { return F_ATOM_ERROR; }
+  const bool ok = UNPADDED ? atomparsing::is_valid_false_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_false_atom(value);
+  if (!ok) { return F_ATOM_ERROR; }
   tape.append(0, internal::tape_type::FALSE_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("false");
   if (!atomparsing::is_valid_false_atom(value, iter.remaining_len())) { return F_ATOM_ERROR; }
   tape.append(0, internal::tape_type::FALSE_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("null");
-  if (!atomparsing::is_valid_null_atom(value)) { return N_ATOM_ERROR; }
+  const bool ok = UNPADDED ? atomparsing::is_valid_null_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_null_atom(value);
+  if (!ok) { return N_ATOM_ERROR; }
   tape.append(0, internal::tape_type::NULL_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("null");
   if (!atomparsing::is_valid_null_atom(value, iter.remaining_len())) { return N_ATOM_ERROR; }
   tape.append(0, internal::tape_type::NULL_VALUE);
   return SUCCESS;
 }

+#if SIMDJSON_ENABLE_NAN_INF
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept {
+  iter.log_value("nan");
+  // For unpadded input use the length-aware validator so the 'infinity'-style
+  // 8-byte compare cannot read past the buffer on a malformed token at the end.
+  const bool ok = UNPADDED ? atomparsing::is_valid_nan_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_nan_atom(value);
+  if (!ok) { return errc; }
+  tape.append_double(std::numeric_limits<double>::quiet_NaN());
+  return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept {
+  iter.log_value("nan");
+  if (!atomparsing::is_valid_nan_atom(value, iter.remaining_len())) { return errc; }
+  tape.append_double(std::numeric_limits<double>::quiet_NaN());
+  return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_inf_atom(json_iterator &iter, const uint8_t *value) noexcept {
+  iter.log_value("inf");
+  // Because 'inf' is an extension, non a canonical atom, a tape error should be returned on failure.
+  // For unpadded input use the length-aware validator so the 'infinity' 8-byte
+  // compare cannot read past the buffer on a malformed token at the end.
+  const bool ok = UNPADDED ? atomparsing::is_valid_inf_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_inf_atom(value);
+  if (!ok) { return TAPE_ERROR; }
+  tape.append_double(std::numeric_limits<double>::infinity());
+  return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_inf_atom(json_iterator &iter, const uint8_t *value) noexcept {
+  iter.log_value("inf");
+  // Because 'inf' is an extension, non a canonical atom, a tape error should be returned on failure
+  if (!atomparsing::is_valid_inf_atom(value, iter.remaining_len())) { return TAPE_ERROR; }
+  tape.append_double(std::numeric_limits<double>::infinity());
+  return SUCCESS;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
 // private:

-simdjson_inline uint32_t tape_builder::next_tape_index(json_iterator &iter) const noexcept {
+template <bool UNPADDED>
+simdjson_inline uint32_t tape_builder_impl<UNPADDED>::next_tape_index(json_iterator &iter) const noexcept {
   return uint32_t(tape.next_tape_loc - iter.dom_parser.doc->tape.get());
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
   auto start_index = next_tape_index(iter);
   tape.append(start_index+2, start);
   tape.append(start_index, end);
   return SUCCESS;
 }

-simdjson_inline void tape_builder::start_container(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_inline void tape_builder_impl<UNPADDED>::start_container(json_iterator &iter) noexcept {
   iter.dom_parser.open_containers[iter.depth].tape_index = next_tape_index(iter);
   iter.dom_parser.open_containers[iter.depth].count = 0;
   tape.skip(); // We don't actually *write* the start element until the end.
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
   // Write the ending tape element, pointing at the start location
   const uint32_t start_tape_index = iter.dom_parser.open_containers[iter.depth].tape_index;
   tape.append(start_tape_index, end);
@@ -42077,13 +52990,15 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::end_container(json
   return SUCCESS;
 }

-simdjson_inline uint8_t *tape_builder::on_start_string(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_inline uint8_t *tape_builder_impl<UNPADDED>::on_start_string(json_iterator &iter) noexcept {
   // we advance the point, accounting for the fact that we have a NULL termination
   tape.append(current_string_buf_loc - iter.dom_parser.doc->string_buf.get(), internal::tape_type::STRING);
   return current_string_buf_loc + sizeof(uint32_t);
 }

-simdjson_inline void tape_builder::on_end_string(uint8_t *dst) noexcept {
+template <bool UNPADDED>
+simdjson_inline void tape_builder_impl<UNPADDED>::on_end_string(uint8_t *dst) noexcept {
   uint32_t str_length = uint32_t(dst - (current_string_buf_loc + sizeof(uint32_t)));
   // TODO check for overflow in case someone has a crazy string (>=4GB?)
   // But only add the overflow check when the document itself exceeds 4GB
@@ -42390,10 +53305,6 @@ simdjson_inline int count_ones(uint64_t input_num) {
   return __lasx_xvpickve2gr_w(__lasx_xvpcnt_d(__m256i(v4u64{input_num, 0, 0, 0})), 0);
 }

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2, uint64_t *result) {
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-}

 } // unnamed namespace
 } // namespace lasx
@@ -43140,6 +54051,9 @@ namespace atomparsing {
 // to the compile-time constant 1936482662.
 simdjson_inline uint32_t string_to_uint32(const char* str) { uint32_t val; std::memcpy(&val, str, sizeof(uint32_t)); return val; }

+// Acts on the same principle as string_to_uint32, but on an 8-byte block of memory
+simdjson_inline uint64_t string_to_uint64(const char* str) { uint64_t val; std::memcpy(&val, str, sizeof(uint64_t)); return val; }
+

 // Again in str4ncmp we use a memcpy to avoid undefined behavior. The memcpy may appear expensive.
 // Yet all decent optimizing compilers will compile memcpy to a single instruction, just about.
@@ -43151,6 +54065,28 @@ simdjson_inline uint32_t str4ncmp(const uint8_t *src, const char* atom) {
   return srcval ^ string_to_uint32(atom);
 }

+// Checks that the first 8 characters of the input string match the given atom in a case-insensitive manner.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint64_t str8ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+  uint64_t srcval; // we want to avoid unaligned 32-bit loads (undefined in C/C++)
+  static_assert(sizeof(uint64_t) <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be larger than 8 bytes");
+  std::memcpy(&srcval, src, sizeof(uint64_t));
+
+  return (srcval | 0x2020202020202020ull) ^ string_to_uint64(atom);
+}
+
+// Checks that the first 3 characters of 'src' match 'atom' in a case-insensitive way.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint32_t str3ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+  return ((src[0] | 0x20) ^ atom[0]) //
+       | ((src[1] | 0x20) ^ atom[1]) //
+       | ((src[2] | 0x20) ^ atom[2]);
+}
+
 simdjson_warn_unused
 simdjson_inline bool is_valid_true_atom(const uint8_t *src) {
   return (str4ncmp(src, "true") | jsoncharutils::is_not_structural_or_whitespace(src[4])) == 0;
@@ -43187,6 +54123,71 @@ simdjson_inline bool is_valid_null_atom(const uint8_t *src, size_t len) {
   else { return false; }
 }

+#if SIMDJSON_ENABLE_NAN_INF
+// "nan" is 3 bytes; we check characters and then verify the next
+// character is structural or whitespace. We accept both "nan" and "NaN".
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src) {
+  return (str3ncmp_case_insensitive(src, "nan")
+        | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+}
+
+// checks that the next four characters of a string are 'nan"', where the 'nan'
+// is checked in a case-insensitive way.
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_in_string(const uint8_t *src) {
+  return (str3ncmp_case_insensitive(src, "nan") | (src[3] ^ '"')) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src, size_t len) {
+  if (len > 3) { return is_valid_nan_atom(src); }
+  if (len == 3) { return str3ncmp_case_insensitive(src, "nan") == 0; }
+  return false;
+}
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src) {
+  bool is_short_inf = (str3ncmp_case_insensitive(src, "inf")
+                     | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+  if(is_short_inf) return true;
+
+  // Check for 'infinity' (any capitalization)
+  return (str8ncmp_case_insensitive(src, "infinity") | jsoncharutils::is_not_structural_or_whitespace(src[8])) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_in_string(const uint8_t *src) {
+  bool is_short_inf = (str3ncmp_case_insensitive(src, "inf") | (src[3] ^ '"')) == 0;
+  if(is_short_inf) return true;
+
+  return (str8ncmp_case_insensitive(src, "infinity") | (src[8] ^ '"')) == 0;
+}
+
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src, size_t len) {
+  if (len > 8) { return is_valid_inf_atom(src); }
+  if (len == 8 && str8ncmp_case_insensitive(src, "infinity") == 0) {
+    return true;
+  }
+  if (len > 3) {
+    return (str3ncmp_case_insensitive(src, "inf")
+          | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+  }
+  if (len == 3) { return str3ncmp_case_insensitive(src, "inf") == 0; }
+  return false;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
 } // namespace atomparsing
 } // unnamed namespace
 } // namespace lasx
@@ -43277,7 +54278,7 @@ inline simdjson_warn_unused error_code dom_parser_implementation::set_capacity(s
 }

 inline simdjson_warn_unused error_code dom_parser_implementation::set_max_depth(size_t max_depth) noexcept {
-  if(max_depth == 0 || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
+  if(max_depth == 0 || max_depth > SIMDJSON_MAX_DEPTH || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
   // Stage 2 stacks
   open_containers.reset(new (std::nothrow) open_container[max_depth]);
   is_array.reset(new (std::nothrow) bool[max_depth]);
@@ -43456,6 +54457,7 @@ protected:
 /* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_NUMBERPARSING_H */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/base.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/jsoncharutils.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/atomparsing.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/internal/numberparsing_tables.h" */
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */

@@ -43495,6 +54497,18 @@ simdjson_inline double to_double(uint64_t mantissa, uint64_t real_exponent, bool
     return d;
 }

+// Convert a mantissa, an exponent and a sign bit into an ieee32 float (binary32).
+// The real_exponent needs to be in [0, 254] (technically real_exponent = 255 would be acceptable).
+// The mantissa should be in [0,1<<24). The bit at index (1U << 23) will be zeroed.
+simdjson_inline float to_float(uint32_t mantissa, uint32_t real_exponent, bool negative) {
+    float f;
+    mantissa &= ~(uint32_t(1) << 23);
+    mantissa |= real_exponent << 23;
+    mantissa |= ((static_cast<uint32_t>(negative)) << 31);
+    std::memcpy(&f, &mantissa, sizeof(f));
+    return f;
+}
+
 // Attempts to compute i * 10^(power) exactly; and if "negative" is
 // true, negate the result.
 // This function will only work in some cases, when it does not work, success is
@@ -43751,6 +54765,219 @@ simdjson_inline bool compute_float_64(int64_t power, uint64_t i, bool negative,
   return true;
 }

+// Attempts to compute i * 10^(power) as a binary32 (float) value; and if
+// "negative" is true, negate the result.
+//
+// This is the same Eisel-Lemire algorithm as compute_float_64, retargeted at
+// binary32. It is adapted from fast_float
+// (https://github.com/fastfloat/fast_float), where the routine is written once
+// and instantiated for each binary format. We only need 24 bits of mantissa
+// instead of 53, so we ask the truncated product for 24 + 2 = 26 accurate bits
+// (plus one bit that may be lost to the "upperbit" shift), and we shift the
+// product right by (upperbit + 64 - 23 - 3) instead of (upperbit + 64 - 52 - 3).
+//
+// The power-of-five table is shared with the binary64 code: the useful range
+// for binary32, [smallest_power_binary32, largest_power_binary32], is a strict
+// subset of [smallest_power, largest_power].
+//
+// The function returns false when the result would be infinite: simdjson
+// refuses to parse infinite values, so the caller reports an error. Unlike
+// compute_float_64, a false return never means "try harder": the accuracy of
+// the two-limb product is guaranteed by Noble Mushtak and Daniel Lemire,
+// "Fast Number Parsing Without Fallback" (https://arxiv.org/abs/2212.06644).
+// The caller must still fall back to a slow path when the 64-bit mantissa i
+// was truncated (more than 19 significant digits).
+//
+// We assume that power is in the [smallest_power, largest_power] interval: the
+// caller is responsible for this check.
+simdjson_inline bool compute_float_32(int64_t power, uint64_t i, bool negative, float &d) {
+  // Powers of ten that are exactly representable as binary32 values: 10^k is
+  // exact as long as k <= 10 (5^10 = 9765625 < 2**24 while 5^11 > 2**24).
+  static constexpr float power_of_ten_float[] = {
+      1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+      1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+  // The range of powers of ten that a non-zero, finite binary32 value can be
+  // built from. Anything smaller rounds to zero, anything larger is infinite.
+  // These are the smallest_power_of_ten()/largest_power_of_ten() constants that
+  // fast_float uses for binary32.
+  constexpr int smallest_power_binary32 = -65;
+  constexpr int largest_power_binary32 = 38;
+
+  // We start with the fast path described in
+  // Clinger WD. How to read floating point numbers accurately.
+  // ACM SIGPLAN Notices. 1990
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+  // We cannot be certain that x/y is rounded to nearest.
+  if (0 <= power && power <= 10 && i <= 16777215)
+#else
+  if (-10 <= power && power <= 10 && i <= 16777215)
+#endif
+  {
+    // Convert the integer into a float. This is lossless since
+    // 0 <= i <= 2^24 - 1.
+    d = float(i);
+    // Both d and the power of ten are exactly representable as binary32
+    // values, so the product (or quotient) is correctly rounded.
+    if (power < 0) {
+      d = d / power_of_ten_float[-power];
+    } else {
+      d = d * power_of_ten_float[power];
+    }
+    if (negative) {
+      d = -d;
+    }
+    return true;
+  }
+
+  // The fast path has failed, so we fall back on the Eisel-Lemire algorithm.
+  // It needs i > 0 (so that the leading bit of i can be normalized), so we
+  // handle i == 0 separately. We also handle the powers of ten that are so
+  // small (or so large) that the answer is zero (or infinite) whatever the
+  // mantissa is.
+  if (i == 0 || power < smallest_power_binary32) {
+    d = negative ? -0.0f : 0.0f;
+    return true;
+  }
+  if (power > largest_power_binary32) {
+    // We have, for sure, an infinite value.
+    return false;
+  }
+
+  // The exponent is 128 + 63 + power + floor(log(5**power)/log(2)).
+  // The 128 comes from the ieee32 standard (the minimal exponent is -127).
+  // The 63 comes from the fact that we use a 64-bit word.
+  // See compute_float_64 for a discussion of the magical 152170 + 65536.
+  int64_t exponent = (((152170 + 65536) * power) >> 16) + 128 + 63;
+
+  // We want the most significant bit of i to be 1. Shift if needed.
+  int lz = leading_zeroes(i);
+  i <<= lz;
+
+  // We want the most significant 64 bits of the product i * 5**power. It is
+  // safe to index the table because
+  // smallest_power <= smallest_power_binary32 <= power
+  //               <= largest_power_binary32 <= largest_power.
+  const uint32_t index = 2 * uint32_t(power - simdjson::internal::smallest_power);
+#if SIMDJSON_STATIC_REFLECTION
+  simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index]);
+#else
+  simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index]);
+#endif
+
+  // Unless the least significant 38 bits of the high (64-bit) part of the full
+  // product are all 1s, then we know that the most significant 26 bits are
+  // exact and no further work is needed. Having 26 bits is necessary because
+  // we need 24 bits for the mantissa but we have to have one rounding bit and
+  // we can waste a bit if the most significant bit of the product is zero.
+  // (For binary64, the same reasoning gives 55 bits and a 9-bit mask.)
+  if((firstproduct.high & 0x3FFFFFFFFF) == 0x3FFFFFFFFF) {
+    // The truncated multiplication was not accurate enough; use the next 64
+    // bits of the power of five to refine it. See compute_float_64 for a
+    // detailed discussion.
+#if SIMDJSON_STATIC_REFLECTION
+    simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index + 1]);
+#else
+    simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index + 1]);
+#endif
+    firstproduct.low += secondproduct.high;
+    if(secondproduct.high > firstproduct.low) { firstproduct.high++; }
+  }
+  uint64_t lower = firstproduct.low;
+  uint64_t upper = firstproduct.high;
+  // The final mantissa should be 24 bits with a leading 1.
+  // We shift it so that it occupies 25 bits with a leading 1.
+  ///////
+  uint64_t upperbit = upper >> 63;
+  uint64_t mantissa = upper >> (upperbit + 38); // 38 == 64 - 23 - 3
+  lz += int(1 ^ upperbit);
+
+  // Here we have mantissa < (1<<25).
+  int64_t real_exponent = exponent - lz;
+  if (simdjson_unlikely(real_exponent <= 0)) { // we have a subnormal?
+    // Here we have that real_exponent <= 0 so -real_exponent >= 0
+    if(-real_exponent + 1 >= 64) { // if we have more than 64 bits below the minimum exponent, you have a zero for sure.
+      d = negative ? -0.0f : 0.0f;
+      return true;
+    }
+    // next line is safe because -real_exponent + 1 < 64
+    mantissa >>= -real_exponent + 1;
+    // Thankfully, we can't have both "round-to-even" and subnormals because
+    // "round-to-even" only occurs for powers close to 0.
+    mantissa += (mantissa & 1); // round up
+    mantissa >>= 1;
+    // As in compute_float_64, rounding up may take us out of the subnormal
+    // range, so we can only decide after rounding.
+    real_exponent = (mantissa < (uint64_t(1) << 23)) ? 0 : 1;
+    d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+    return true;
+  }
+  // We have to round to even. The "to even" part is only a problem when we are
+  // right in between two floats, which we guard against. The bounds on the
+  // power of ten are those of fast_float's
+  // min_exponent_round_to_even()/max_exponent_round_to_even() for binary32:
+  // when q >= 0, (2m+1) must be divisible by 5^q with 5^q <= 2^25, so q <= 10;
+  // when q < 0, we need 2^24 x 5^{-q} < 2^{64}, so q >= -17.
+  if (simdjson_unlikely((lower <= 1) && (power >= -17) && (power <= 10) && ((mantissa & 3) == 1))) {
+    if((mantissa << (upperbit + 38)) == upper) {
+      mantissa &= ~uint64_t(1);   // flip it so that we do not round up
+    }
+  }
+
+  mantissa += mantissa & 1;
+  mantissa >>= 1;
+
+  // Here we have mantissa < (1<<24), unless there was an overflow
+  if (mantissa >= (uint64_t(1) << 24)) {
+    mantissa = (uint64_t(1) << 23);
+    real_exponent++;
+  }
+  mantissa &= ~(uint64_t(1) << 23);
+  // we have to check that real_exponent is in range, otherwise we bail out
+  if (simdjson_unlikely(real_exponent > 254)) {
+    // We have an infinite value!!! We could actually throw an error here if we could.
+    return false;
+  }
+  d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+  return true;
+}
+
+#if SIMDJSON_ENABLE_NAN_INF
+// Parses a nan or infinity. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, double& d) noexcept {
+  if (atomparsing::is_valid_inf_atom(src)) {
+    double inf = std::numeric_limits<double>::infinity();
+    d = negative ? -inf : inf;
+    return true;
+  }
+
+  if (atomparsing::is_valid_nan_atom(src)) {
+    d = std::numeric_limits<double>::quiet_NaN();
+    return true;
+  }
+
+  return false;
+}
+
+// Parses a nan or infinity as a binary32 value. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, float& d) noexcept {
+  if (atomparsing::is_valid_inf_atom(src)) {
+    float inf = std::numeric_limits<float>::infinity();
+    d = negative ? -inf : inf;
+    return true;
+  }
+
+  if (atomparsing::is_valid_nan_atom(src)) {
+    d = std::numeric_limits<float>::quiet_NaN();
+    return true;
+  }
+
+  return false;
+}
+#endif
+
 // We call a fallback floating-point parser that might be slow. Note
 // it will accept JSON numbers, but the JSON spec. is more restrictive so
 // before you call parse_float_fallback, you need to have validated the input
@@ -43788,6 +55015,16 @@ static bool parse_float_fallback(const uint8_t *ptr, const uint8_t *end_ptr, dou
   return !(*outDouble > (std::numeric_limits<double>::max)() || *outDouble < std::numeric_limits<double>::lowest());
 }

+// Same as parse_float_fallback, but for binary32 (float) values. Going through
+// the binary64 fallback and then rounding to binary32 would be subject to
+// double rounding, so we run the fallback algorithm directly on binary32.
+static bool parse_float_fallback(const uint8_t *ptr, float *outFloat) {
+  *outFloat = simdjson::internal::from_chars_float(reinterpret_cast<const char *>(ptr));
+  // We do not accept infinite values. See the binary64 version above for why we
+  // do not use std::isfinite.
+  return !(*outFloat > (std::numeric_limits<float>::max)() || *outFloat < std::numeric_limits<float>::lowest());
+}
+
 // check quickly whether the next 8 chars are made of digits
 // at a glance, it looks better than Mula's
 // http://0x80.pl/articles/swar-digits-validate.html
@@ -43806,6 +55043,28 @@ simdjson_inline bool is_made_of_eight_digits_fast(const uint8_t *chars) {
           0x3333333333333333);
 }

+// The same idea for four characters. A block of eight only bites when eight
+// digits are left, so a run of 4 to 7 digits used to fall back to one digit at a
+// time; taking four of them at once is what makes the tail of a long fraction
+// cheap. Adapted from fast_float, credit @aqrit.
+simdjson_inline bool is_made_of_four_digits_fast(const uint8_t *chars) {
+  uint32_t val;
+  // Reads up to 3 bytes beyond the digits, which SIMDJSON_PADDING covers.
+  static_assert(3 <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be bigger than 3");
+  std::memcpy(&val, chars, 4);
+  return !(((val + 0x46464646) | (val - 0x30303030)) & 0x80808080);
+}
+
+// Only call this when is_made_of_four_digits_fast() says the four characters are
+// digits. Little-endian: guarded by SIMDJSON_SWAR_NUMBER_PARSING.
+simdjson_inline uint32_t parse_four_digits_unrolled(const uint8_t *chars) {
+  uint32_t val;
+  std::memcpy(&val, chars, 4);
+  val -= 0x30303030;
+  val = (val * 10) + (val >> 8);
+  return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
 template<typename I>
 SIMDJSON_NO_SANITIZE_UNDEFINED // We deliberately allow overflow here and check later
 simdjson_inline bool parse_digit(const uint8_t c, I &i) {
@@ -43822,26 +55081,73 @@ simdjson_inline bool is_digit(const uint8_t c) {
   return static_cast<uint8_t>(c - '0') <= 9;
 }

-simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
-  // we continue with the fiction that we have an integer. If the
-  // floating point number is representable as x * 10^z for some integer
-  // z that fits in 53 bits, then we will be able to convert back the
-  // the integer into a float in a lossless manner.
-  const uint8_t *const first_after_period = p;
+// Consumes a run of digits into i, eight at a time while eight are available,
+// then four, then one at a time. Overflow is deliberate: the caller counts the
+// digits and falls back when there are too many for a 64-bit mantissa.
+//
+// Long runs of digits are what a float-heavy document is made of: the fraction
+// of a binary64 printed to full precision is 15 to 17 digits, and reading those
+// one at a time is the single largest cost in parse_double(). Integer parts are
+// left alone: they are usually a handful of digits, and there the check for a
+// block of eight is wasted work.
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_fraction_digits(const uint8_t *&p, uint64_t &i) {
+#ifdef SIMDJSON_SWAR_NUMBER_PARSING
+#if SIMDJSON_SWAR_NUMBER_PARSING
+  while (is_made_of_eight_digits_fast(p)) {
+    i = i * 100000000 + parse_eight_digits_unrolled(p);
+    p += 8;
+  }
+  // A 4 to 7 digit remainder is the common case once the blocks of eight are
+  // gone: 15 fraction digits leave 7, and 17 leave 1 after two blocks.
+  if (is_made_of_four_digits_fast(p)) {
+    i = i * 10000 + parse_four_digits_unrolled(p);
+    p += 4;
+  }
+#endif // SIMDJSON_SWAR_NUMBER_PARSING
+#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
+  while (parse_digit(*p, i)) { p++; }
+}
+
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_float_integer_digits(const uint8_t *&p, uint64_t &i,
+                                                bool &leading_zero) {
+  if (!parse_digit(*p, i)) { leading_zero = true; return; }
+  p++;
+  leading_zero = (i == 0);
+  while (parse_digit(*p, i)) { p++; }
+}

+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_integer_digits(const uint8_t *&p, uint64_t &i) {
 #ifdef SIMDJSON_SWAR_NUMBER_PARSING
 #if SIMDJSON_SWAR_NUMBER_PARSING
-  // this helps if we have lots of decimals!
-  // this turns out to be frequent enough.
+  // Identifiers, timestamps and counters often have eight digits or more.
+  // Prior related work: jsonifier parses integers as eight-digit SWAR words
+  // (str_to_i.hpp, https://github.com/nihilai-collective/Jsonifier). This
+  // takes one such word with parse_eight_digits_unrolled, the routine
+  // simdjson uses for long fractions.
   if (is_made_of_eight_digits_fast(p)) {
     i = i * 100000000 + parse_eight_digits_unrolled(p);
     p += 8;
   }
+  const uint8_t *const swar_end = p + 8;
+  while (p < swar_end && is_made_of_four_digits_fast(p)) {
+    i = i * 10000 + parse_four_digits_unrolled(p);
+    p += 4;
+  }
 #endif // SIMDJSON_SWAR_NUMBER_PARSING
 #endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
-  // Unrolling the first digit makes a small difference on some implementations (e.g. westmere)
-  if (parse_digit(*p, i)) { ++p; }
   while (parse_digit(*p, i)) { p++; }
+}
+
+simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
+  // we continue with the fiction that we have an integer. If the
+  // floating point number is representable as x * 10^z for some integer
+  // z that fits in 53 bits, then we will be able to convert back the
+  // the integer into a float in a lossless manner.
+  const uint8_t *const first_after_period = p;
+  parse_fraction_digits(p, i);
   exponent = first_after_period - p;
   // Decimal without digits (123.) is illegal
   if (exponent == 0) {
@@ -43930,7 +55236,7 @@ simdjson_inline size_t significant_digits(const uint8_t * start_digits, size_t d
 } // unnamed namespace

 /** @private */
-static error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
+inline error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
   if (parse_float_fallback(src, answer)) {
     return SUCCESS;
   }
@@ -44013,15 +55319,17 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
   return SUCCESS;              // always succeeds
 }

-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept  { return false; }
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept  { return false; }
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept { return number_type::signed_integer; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline bool is_negative(const uint8_t *) noexcept  { return false; }
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t *) noexcept  { return false; }
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t *) noexcept { return number_type::signed_integer; }
 #else

 // parse the number at src
@@ -44052,7 +55360,21 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
   size_t digit_count = size_t(p - start_digits);
-  if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) { return INVALID_NUMBER(src); }
+  if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // By this point, we know that our input does not begin with a digit. We will attempt
+    // to handle NaN/Infinity.
+
+    double d;
+    if (compute_nan_inf(p, negative, d)) {
+      writer.append_double(d);
+      return SUCCESS;
+    }
+#endif
+
+    return INVALID_NUMBER(src);
+  }

   //
   // Handle floats if there is a . or e (or both)
@@ -44101,7 +55423,7 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
     // - Therefore, if the number is positive and lower than that, it's overflow.
     // - The value we are looking at is less than or equal to INT64_MAX.
     //
-    }  else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return INVALID_NUMBER(src); }
+    }  else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return BIGINT_NUMBER(src); }
   }

   // Write unsigned if it does not fit in a signed integer.
@@ -44200,7 +55522,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const u
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -44298,7 +55620,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_stri
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -44353,7 +55675,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uin
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -44439,7 +55761,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = src;
   uint64_t i = 0;
-  while (parse_digit(*src, i)) { src++; }
+  parse_integer_digits(src, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -44479,11 +55801,20 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   //
   uint64_t i = 0;
   const uint8_t *p = src;
-  p += parse_digit(*p, i);
-  bool leading_zero = (i == 0);
-  while (parse_digit(*p, i)) { p++; }
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
   // no integer digits, or 0123 (zero must be solo)
-  if ( p == src ) { return INCORRECT_TYPE; }
+  if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no loading digits, the number may be nan or infinity.
+    // Attempt to compute those, and return on success.
+    double d;
+    if (compute_nan_inf(p, negative, d)) { return d; }
+#endif
+
+    return INCORRECT_TYPE;
+  }
   if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }

   //
@@ -44494,9 +55825,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   if (simdjson_likely(*p == '.')) {
     p++;
     const uint8_t *start_decimal_digits = p;
-    if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
-    p++;
-    while (parse_digit(*p, i)) { p++; }
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
     exponent = -(p - start_decimal_digits);

     // Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -44545,6 +55875,97 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   return d;
 }

+// Parse a JSON number into a binary32 (float) value.
+//
+// This mirrors parse_double, but it rounds to binary32 directly instead of
+// rounding to binary64 and then to binary32: the latter is subject to double
+// rounding and would not always produce the nearest float.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * src) noexcept {
+  //
+  // Check for minus sign
+  //
+  bool negative = (*src == '-');
+  src += uint8_t(negative);
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, the number may be nan or infinity.
+    // Attempt to compute those, and return on success.
+    float f;
+    if (compute_nan_inf(p, negative, f)) { return f; }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = p-start_digits > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if (*p == 'e' || *p == 'E') {
+    p++;
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while (parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  float d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_32(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
 simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
   return (*src == '-');
 }
@@ -44697,11 +56118,25 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   //
   uint64_t i = 0;
   const uint8_t *p = src;
-  p += parse_digit(*p, i);
-  bool leading_zero = (i == 0);
-  while (parse_digit(*p, i)) { p++; }
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
   // no integer digits, or 0123 (zero must be solo)
-  if ( p == src ) { return INCORRECT_TYPE; }
+  if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, attempt to parse numbers that are either
+    // NaN or Infinity
+    if (atomparsing::is_valid_inf_in_string(src)) {
+      double inf = std::numeric_limits<double>::infinity();
+      return negative ? -inf : inf;
+    }
+
+    if (atomparsing::is_valid_nan_in_string(src)) {
+      return std::numeric_limits<double>::quiet_NaN();
+    }
+#endif
+
+    return INCORRECT_TYPE;
+  }
   if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }

   //
@@ -44712,9 +56147,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   if (simdjson_likely(*p == '.')) {
     p++;
     const uint8_t *start_decimal_digits = p;
-    if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
-    p++;
-    while (parse_digit(*p, i)) { p++; }
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
     exponent = -(p - start_decimal_digits);

     // Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -44763,6 +56197,99 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   return d;
 }

+// Parse a JSON number held inside a JSON string into a binary32 (float) value.
+// See parse_float for why we do not simply round parse_double_in_string.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * src) noexcept {
+  //
+  // Check for minus sign
+  //
+  bool negative = (*(src + 1) == '-');
+  src += uint8_t(negative) + 1;
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, attempt to parse numbers that are either
+    // NaN or Infinity
+    if (atomparsing::is_valid_inf_in_string(src)) {
+      float inf = std::numeric_limits<float>::infinity();
+      return negative ? -inf : inf;
+    }
+
+    if (atomparsing::is_valid_nan_in_string(src)) {
+      return std::numeric_limits<float>::quiet_NaN();
+    }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = p-start_digits > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if (*p == 'e' || *p == 'E') {
+    p++;
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while (parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if (*p != '"') { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  float d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_32(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
 } // unnamed namespace
 #endif // SIMDJSON_SKIPNUMBERPARSING

@@ -45061,10 +56588,6 @@ simdjson_inline int count_ones(uint64_t input_num) {
   return __lasx_xvpickve2gr_w(__lasx_xvpcnt_d(__m256i(v4u64{input_num, 0, 0, 0})), 0);
 }

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2, uint64_t *result) {
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-}

 } // unnamed namespace
 } // namespace lasx
@@ -46608,6 +58131,296 @@ simdjson_inline uint32_t find_next_document_index(dom_parser_implementation &par
   return 0;
 }

+/**
+ * Sentinel value returned to indicate a document started but didn't fit
+ * (CAPACITY error), as opposed to 0 which means no document content found
+ * (EMPTY).
+ */
+constexpr uint32_t DOCUMENT_TOO_LARGE = UINT32_MAX;
+
+/**
+ * For RFC 7464 JSON text sequences, filter RS from structural indexes and
+ * find batch boundaries.
+ *
+ * In JSON sequence mode, RS (0x1E) marks the start of each JSON text.
+ * RS bytes appear in structural_indexes as they are classified as scalars.
+ * This function:
+ * 1. Scans structural_indexes to find and count RS positions
+ * 2. Filters RS out of structural_indexes in-place
+ * 3. Determines batch boundaries based on RS positions
+ *
+ * @param parser The parser with structural_indexes and buf.
+ * @param len The length of the current batch buffer.
+ * @param is_final True if this is the final batch (no more data coming).
+ * @param next_batch_start Output: offset where the next batch should start.
+ * @param scan_len Offset past which no value start may be derived. Equals len
+ *        unless stage 1 dropped a trailing unclosed string, whose bytes it
+ *        never validated.
+ * @return The number of structural indexes to keep (after RS filtering),
+ *         0 if no document content found (EMPTY),
+ *         or DOCUMENT_TOO_LARGE if a document started but didn't fit (CAPACITY).
+ */
+simdjson_inline uint32_t find_next_document_index_json_sequence(
+    dom_parser_implementation &parser,
+    size_t len,
+    bool is_final,
+    uint32_t &next_batch_start,
+    size_t scan_len) {
+  // Default: next batch starts at end of buffer
+  next_batch_start = uint32_t(len);
+
+  if (parser.n_structural_indexes == 0) { return 0; }
+
+  // Phase 1: Scan structural_indexes to find RS positions and handle them.
+  // RS marks the start of a JSON text. For objects/arrays, the '{' or '[' after RS
+  // is already in structural_indexes (it's an operator). For scalars like numbers,
+  // the digit following RS is NOT in structural_indexes because the scanner sees
+  // RS as a scalar, making the digit a scalar continuation, not a start.
+  // We must: (1) remove RS from structural_indexes, and (2) for scalars, add the
+  // actual value start position.
+  // The EOF sentinel: len, or where a discarded unclosed string starts.
+  const uint32_t sentinel = parser.structural_indexes[parser.n_structural_indexes];
+  uint32_t write_idx = 0;
+  uint32_t last_rs_pos = 0;
+  uint32_t rs_count = 0;
+
+  for (uint32_t read_idx = 0; read_idx < parser.n_structural_indexes; read_idx++) {
+    const uint32_t pos = parser.structural_indexes[read_idx];
+    if (parser.buf[pos] == 0x1E) {
+      // This is an RS character - find the actual JSON value start.
+      last_rs_pos = pos;
+      rs_count++;
+      // Skip past this RS and any whitespace *and any additional RSes*
+      // to locate the real value. Consecutive RSes are degenerate
+      // "empty records" per RFC 7464; we collapse them here. They do
+      // not always appear as separate entries in structural_indexes
+      // because the scanner groups runs of adjacent non-whitespace
+      // scalar bytes (including RS) into a single scalar start.
+      uint32_t value_start = pos + 1;
+      while (value_start < scan_len) {
+        const uint8_t c = parser.buf[value_start];
+        if (c == ' ' || c == '\t' || c == '\n' || c == '\r') {
+          value_start++;
+        } else if (c == 0x1E) {
+          // Collapsed empty record. Still count it so rs_count reflects
+          // the true number of record markers and last_rs_pos tracks
+          // the final one.
+          last_rs_pos = value_start;
+          rs_count++;
+          value_start++;
+        } else {
+          break;
+        }
+      }
+      // If the scanner emitted additional structurals inside the
+      // whitespace+RS run we just walked over (i.e., isolated RSes
+      // separated by whitespace), skip past them so we do not
+      // double-count or double-emit.
+      while (read_idx + 1 < parser.n_structural_indexes &&
+             parser.structural_indexes[read_idx + 1] < value_start) {
+        read_idx++;
+      }
+      // Check if the value start is an operator (always present in
+      // scanner structural_indexes) or a scalar-like start (which may
+      // be missing from structural_indexes and must be added here).
+      // Note: '"' is NOT always in structural_indexes. The scanner
+      // classifies '"' as a scalar character and emits it as a
+      // structural only when it is a *scalar start* (preceded by
+      // whitespace or an operator). When '"' immediately follows an
+      // RS (which the scanner also classifies as scalar), it is
+      // treated as a scalar continuation and not emitted - so we
+      // must add it here just like any other scalar value.
+      if (value_start < scan_len) {
+        const uint8_t c = parser.buf[value_start];
+        const bool is_operator =
+            (c == '{' || c == '}' || c == '[' || c == ']' ||
+             c == ':' || c == ',');
+        // If the next scanner structural is exactly at value_start,
+        // the scanner already emitted it (it followed whitespace) and
+        // we must not add a duplicate - a subsequent iteration will
+        // copy it into write_idx.
+        const bool already_emitted =
+            (read_idx + 1 < parser.n_structural_indexes &&
+             parser.structural_indexes[read_idx + 1] == value_start);
+        if (!is_operator && !already_emitted) {
+          // Scalar value (number/true/false/null/string) - add its
+          // position since scanner missed it.
+          parser.structural_indexes[write_idx++] = value_start;
+        }
+      }
+    } else {
+      // Not RS, copy to output
+      parser.structural_indexes[write_idx++] = pos;
+    }
+  }
+
+  // Update structural index count. Compaction left a stale index in the slot
+  // past the end: restore the EOF sentinel that stage 1 had planted there, which
+  // document_stream::truncated_bytes() reads after a final batch.
+  parser.n_structural_indexes = write_idx;
+  parser.structural_indexes[write_idx] = sentinel;
+
+  if (parser.n_structural_indexes == 0) {
+    // Only RS markers here: the last one opens a record continuing past the
+    // window, so that is where the next batch resumes.
+    if (!is_final && rs_count > 0) { next_batch_start = last_rs_pos; }
+    return 0;
+  }
+  if (rs_count == 0) {
+    // No RS found; for final batch, try generic boundary detection
+    return is_final ? find_next_document_index(parser) : 0;
+  }
+
+  // Phase 2: Determine batch boundaries based on RS positions
+
+  if (is_final) {
+    // Final batch: all documents are complete (last one ends at EOF).
+    // In json_sequence mode, RS markers define document boundaries, so all
+    // remaining structurals form complete documents. Return them all directly.
+    // (Calling find_next_document_index() would fail for scalar documents.)
+    return parser.n_structural_indexes;
+  }
+
+  // Partial batch: need to find complete documents only.
+  // A document starting at an RS is complete if there is another RS after it.
+  next_batch_start = last_rs_pos;
+
+  if (rs_count < 2) {
+    // Only one RS, so we have at most one document that may be incomplete.
+    // We cannot confirm it is complete without another RS.
+    // Return DOCUMENT_TOO_LARGE if content was found (write_idx > 0), 0 if only separators.
+    return (parser.n_structural_indexes > 0) ? DOCUMENT_TOO_LARGE : 0;
+  }
+
+  // We have at least 2 RS markers. The last complete document ends before last_rs_pos.
+
+  // Find the structural index cutoff: keep only structurals < last_rs_pos.
+  // Since we already filtered RS, all remaining structurals are valid.
+  // We iterate backward to find the last structural before last_rs_pos.
+  uint32_t keep_count = 0;
+  for (uint32_t i = parser.n_structural_indexes; i > 0; i--) {
+    if (parser.structural_indexes[i - 1] < last_rs_pos) {
+      keep_count = i;
+      break;
+    }
+  }
+
+  // No structurals before the last RS - no complete documents
+  if (keep_count == 0) { return 0; }
+
+  // All documents before the last RS are complete by definition (the next RS
+  // confirms their end). No need to call find_next_document_index() which
+  // would fail for scalar documents like `1` or `"hello"`.
+  return keep_count;
+}
+
+/**
+ * Filter comma-delimited documents by removing root-level commas from
+ * structural indexes.
+ *
+ * For comma-delimited format like `{...},{...},{...}`, we need to remove
+ * the commas that separate documents (depth 0) while preserving commas
+ * inside arrays and objects (depth > 0).
+ *
+ * After filtering, the structural indexes look like whitespace-delimited
+ * documents, so find_next_document_index() works unchanged.
+ *
+ * @param parser The parser with structural_indexes and buf.
+ * @param len The length of the current batch buffer.
+ * @param is_final True if this is the final batch (no more data coming).
+ * @param next_batch_start Output: offset where the next batch should start.
+ * @return The number of structural indexes to keep,
+ *         0 if no document content found (EMPTY),
+ *         or DOCUMENT_TOO_LARGE if a document started but didn't fit (CAPACITY).
+ */
+simdjson_inline uint32_t filter_comma_delimited(
+    dom_parser_implementation &parser,
+    size_t len,
+    bool is_final,
+    uint32_t &next_batch_start) {
+  // Default: next batch starts at end of buffer
+  next_batch_start = uint32_t(len);
+
+  if (parser.n_structural_indexes == 0) { return 0; }
+
+  // Track depth to identify root-level commas (depth 0)
+  int depth = 0;
+  // The EOF sentinel: len, or where a discarded unclosed string starts.
+  const uint32_t sentinel = parser.structural_indexes[parser.n_structural_indexes];
+  uint32_t write_idx = 0;
+  uint32_t last_root_comma_pos = 0;
+  uint32_t root_comma_count = 0;
+
+  for (uint32_t i = 0; i < parser.n_structural_indexes; i++) {
+    uint32_t idx = parser.structural_indexes[i];
+    uint8_t c = parser.buf[idx];
+
+    switch (c) {
+      case '{': case '[':
+        depth++;
+        parser.structural_indexes[write_idx++] = idx;
+        break;
+      case '}': case ']':
+        depth--;
+        parser.structural_indexes[write_idx++] = idx;
+        break;
+      case ',':
+        if (depth == 0) {
+          // Root-level comma = document boundary, skip it
+          last_root_comma_pos = idx;
+          root_comma_count++;
+          continue;
+        }
+        parser.structural_indexes[write_idx++] = idx;
+        break;
+      default:
+        // Colons, scalars, etc.
+        parser.structural_indexes[write_idx++] = idx;
+        break;
+    }
+  }
+
+  // Update structural index count. Compaction left a stale index in the slot
+  // past the end: restore the EOF sentinel that stage 1 had planted there, which
+  // document_stream::truncated_bytes() reads after a final batch.
+  parser.n_structural_indexes = write_idx;
+  parser.structural_indexes[write_idx] = sentinel;
+
+  if (parser.n_structural_indexes == 0) { return 0; }
+
+  if (is_final) {
+    // Final batch: use standard boundary detection on filtered indexes
+    return find_next_document_index(parser);
+  }
+
+  // Partial batch: need to find complete documents only.
+  // A document ending with a root comma is complete.
+  if (root_comma_count == 0) {
+    // No root commas found; we cannot confirm any document is complete.
+    // The whole batch might be one incomplete document.
+    // Return DOCUMENT_TOO_LARGE if content was found (write_idx > 0), 0 if only commas.
+    return (parser.n_structural_indexes > 0) ? DOCUMENT_TOO_LARGE : 0;
+  }
+
+  // We have at least one root comma. Documents before the last comma are complete.
+  next_batch_start = last_root_comma_pos + 1;
+
+  // Find the structural index cutoff: keep only structurals < last_root_comma_pos
+  uint32_t keep_count = 0;
+  for (uint32_t i = parser.n_structural_indexes; i > 0; i--) {
+    if (parser.structural_indexes[i - 1] < last_root_comma_pos) {
+      keep_count = i;
+      break;
+    }
+  }
+
+  if (keep_count == 0) { return 0; }
+
+  // Use standard boundary detection on the complete portion
+  parser.n_structural_indexes = keep_count;
+  return find_next_document_index(parser);
+}
+
 } // namespace stage1
 } // unnamed namespace
 } // namespace lasx
@@ -47041,7 +58854,6 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
         return EMPTY;
       }
     }
-
     parser.n_structural_indexes = new_structural_indexes;
   } else if (partial == stage1_mode::streaming_final) {
     if(have_unclosed_string) { parser.n_structural_indexes--; }
@@ -47069,6 +58881,75 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
         // the trailing garbage.
         return EMPTY;
     }
+  } else if (partial == stage1_mode::json_sequence_partial) {
+    // RFC 7464: use RS positions for batch boundaries
+    // A discarded unclosed string also caps how far the filter may scan.
+    size_t scan_len = len;
+    if(have_unclosed_string) {
+      parser.n_structural_indexes--;
+      if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return CAPACITY; }
+      scan_len = parser.structural_indexes[parser.n_structural_indexes];
+    }
+    uint32_t next_batch_start = uint32_t(len);
+    auto new_structural_indexes = find_next_document_index_json_sequence(parser, len, false, next_batch_start, scan_len);
+    if (new_structural_indexes == DOCUMENT_TOO_LARGE) {
+      return CAPACITY;
+    }
+    if (new_structural_indexes == 0) {
+      // An EMPTY batch must still advance next_batch_start, or document_stream
+      // re-parses the same bytes forever. CAPACITY when it cannot.
+      if (next_batch_start == 0) { return CAPACITY; }
+      parser.n_structural_indexes = 0;
+      parser.structural_indexes[0] = next_batch_start;
+      return EMPTY;
+    }
+    parser.n_structural_indexes = new_structural_indexes;
+    parser.structural_indexes[parser.n_structural_indexes] = next_batch_start;
+  } else if (partial == stage1_mode::json_sequence_final) {
+    // RFC 7464: final batch, last document extends to EOF
+    // As above: a discarded unclosed string caps how far the filter may scan.
+    size_t scan_len = len;
+    if(have_unclosed_string) {
+      parser.n_structural_indexes--;
+      scan_len = parser.structural_indexes[parser.n_structural_indexes];
+    }
+    uint32_t next_batch_start = uint32_t(len);
+    parser.n_structural_indexes = find_next_document_index_json_sequence(parser, len, true, next_batch_start, scan_len);
+    // The filter compacted structural_indexes in place and restored the EOF
+    // sentinel past the compacted end, so the copy below is either the start
+    // of a truncated document or len, as in streaming_final.
+    parser.structural_indexes[parser.n_structural_indexes + 1] = parser.structural_indexes[parser.n_structural_indexes];
+    parser.structural_indexes[parser.n_structural_indexes] = uint32_t(len);
+    if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return EMPTY; }
+  } else if (partial == stage1_mode::comma_delimited_partial) {
+    // Comma-delimited: filter root-level commas, use comma positions for batch boundaries
+    if(have_unclosed_string) {
+      parser.n_structural_indexes--;
+      if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return CAPACITY; }
+    }
+    uint32_t next_batch_start = uint32_t(len);
+    auto new_structural_indexes = filter_comma_delimited(parser, len, false, next_batch_start);
+    if (new_structural_indexes == DOCUMENT_TOO_LARGE) {
+      return CAPACITY;
+    }
+    if (new_structural_indexes == 0) {
+      // An EMPTY batch must still advance next_batch_start, or document_stream
+      // re-parses the same bytes forever. CAPACITY when it cannot.
+      if (next_batch_start == 0) { return CAPACITY; }
+      parser.n_structural_indexes = 0;
+      parser.structural_indexes[0] = next_batch_start;
+      return EMPTY;
+    }
+    parser.n_structural_indexes = new_structural_indexes;
+    parser.structural_indexes[parser.n_structural_indexes] = next_batch_start;
+  } else if (partial == stage1_mode::comma_delimited_final) {
+    // Comma-delimited: final batch, last document extends to EOF
+    if(have_unclosed_string) { parser.n_structural_indexes--; }
+    uint32_t next_batch_start = uint32_t(len);
+    parser.n_structural_indexes = filter_comma_delimited(parser, len, true, next_batch_start);
+    parser.structural_indexes[parser.n_structural_indexes + 1] = parser.structural_indexes[parser.n_structural_indexes];
+    parser.structural_indexes[parser.n_structural_indexes] = uint32_t(len);
+    if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return EMPTY; }
   }
   checker.check_eof();
   return checker.errors();
@@ -47151,7 +59032,6 @@ namespace {
 namespace stage2 {

 class json_iterator;
-class structural_iterator;
 struct tape_builder;
 struct tape_writer;

@@ -47448,7 +59328,7 @@ public:
    *
    * - increment_count(iter) - each time a value is found in an array or object.
    */
-  template<bool STREAMING, typename V>
+  template<bool STREAMING, bool UNPADDED, typename V>
   simdjson_warn_unused simdjson_inline error_code walk_document(V &visitor) noexcept;

   /**
@@ -47465,6 +59345,7 @@ public:
    *
    * They may include invalid JSON as well (such as `1.2.3` or `ture`).
    */
+  template<bool UNPADDED>
   simdjson_inline const uint8_t *peek() const noexcept;
   /**
    * Advance to the next token.
@@ -47473,6 +59354,7 @@ public:
    *
    * They may include invalid JSON as well (such as `1.2.3` or `ture`).
    */
+  template<bool UNPADDED>
   simdjson_inline const uint8_t *advance() noexcept;
   /**
    * Get the remaining length of the document, from the start of the current token.
@@ -47521,7 +59403,7 @@ public:
   simdjson_warn_unused simdjson_inline error_code visit_primitive(V &visitor, const uint8_t *value) noexcept;
 };

-template<bool STREAMING, typename V>
+template<bool STREAMING, bool UNPADDED, typename V>
 simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &visitor) noexcept {
   logger::log_start();

@@ -47536,7 +59418,7 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &
   // Read first value
   //
   {
-    auto value = advance();
+    auto value = advance<UNPADDED>();

     // Make sure the outer object or array is closed before continuing; otherwise, there are ways we
     // could get into memory corruption. See https://github.com/simdjson/simdjson/issues/906
@@ -47548,8 +59430,8 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &
     }

     switch (*value) {
-      case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
-      case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+      case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+      case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
       default: SIMDJSON_TRY( visitor.visit_root_primitive(*this, value) ); break;
     }
   }
@@ -47566,29 +59448,29 @@ object_begin:
   SIMDJSON_TRY( visitor.visit_object_start(*this) );

   {
-    auto key = advance();
+    auto key = advance<UNPADDED>();
     if (*key != '"') { log_error("Object does not start with a key"); return TAPE_ERROR; }
     SIMDJSON_TRY( visitor.increment_count(*this) );
     SIMDJSON_TRY( visitor.visit_key(*this, key) );
   }

 object_field:
-  if (simdjson_unlikely( *advance() != ':' )) { log_error("Missing colon after key in object"); return TAPE_ERROR; }
+  if (simdjson_unlikely( *advance<UNPADDED>() != ':' )) { log_error("Missing colon after key in object"); return TAPE_ERROR; }
   {
-    auto value = advance();
+    auto value = advance<UNPADDED>();
     switch (*value) {
-      case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
-      case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+      case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+      case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
       default: SIMDJSON_TRY( visitor.visit_primitive(*this, value) ); break;
     }
   }

 object_continue:
-  switch (*advance()) {
+  switch (*advance<UNPADDED>()) {
     case ',':
       SIMDJSON_TRY( visitor.increment_count(*this) );
       {
-        auto key = advance();
+        auto key = advance<UNPADDED>();
         if (simdjson_unlikely( *key != '"' )) { log_error("Key string missing at beginning of field in object"); return TAPE_ERROR; }
         SIMDJSON_TRY( visitor.visit_key(*this, key) );
       }
@@ -47616,16 +59498,16 @@ array_begin:

 array_value:
   {
-    auto value = advance();
+    auto value = advance<UNPADDED>();
     switch (*value) {
-      case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
-      case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+      case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+      case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
       default: SIMDJSON_TRY( visitor.visit_primitive(*this, value) ); break;
     }
   }

 array_continue:
-  switch (*advance()) {
+  switch (*advance<UNPADDED>()) {
     case ',': SIMDJSON_TRY( visitor.increment_count(*this) ); goto array_value;
     case ']': log_end_value("array"); SIMDJSON_TRY( visitor.visit_array_end(*this) ); goto scope_end;
     default: log_error("Missing comma between array values"); return TAPE_ERROR;
@@ -47653,11 +59535,29 @@ simdjson_inline json_iterator::json_iterator(dom_parser_implementation &_dom_par
     dom_parser{_dom_parser} {
 }

+// Stage 1 leaves a sentinel index of value `len`; reading buf[len] requires
+// padding. For UNPADDED, return a dummy so walk_document can fail cleanly (#2815).
+template<bool UNPADDED>
 simdjson_inline const uint8_t *json_iterator::peek() const noexcept {
-  return &buf[*(next_structural)];
+  const uint32_t idx = *(next_structural);
+  SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+    if (simdjson_unlikely(idx >= dom_parser.len)) {
+      static constexpr uint8_t unpadded_eof_sentinel = 0;
+      return &unpadded_eof_sentinel;
+    }
+  }
+  return &buf[idx];
 }
+template<bool UNPADDED>
 simdjson_inline const uint8_t *json_iterator::advance() noexcept {
-  return &buf[*(next_structural++)];
+  const uint32_t idx = *(next_structural++);
+  SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+    if (simdjson_unlikely(idx >= dom_parser.len)) {
+      static constexpr uint8_t unpadded_eof_sentinel = 0;
+      return &unpadded_eof_sentinel;
+    }
+  }
+  return &buf[idx];
 }
 simdjson_inline size_t json_iterator::remaining_len() const noexcept {
   return dom_parser.len - *(next_structural-1);
@@ -47697,7 +59597,20 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::visit_root_primit
     case '"': return visitor.visit_root_string(*this, value);
     case 't': return visitor.visit_root_true_atom(*this, value);
     case 'f': return visitor.visit_root_false_atom(*this, value);
+#if SIMDJSON_ENABLE_NAN_INF
+    case 'n': {
+      auto err = visitor.visit_root_null_atom(*this, value);
+      if (err == SUCCESS) { return err; }
+      // propagate the error value returned by a bad 'null' atom if parsing 'nan' fails
+      return visitor.visit_root_nan_atom(*this, value, err);
+    }
+    // 'N' isn't a canonically recognized atom, so we return a TAPE_ERROR if failure occurs
+    case 'N': return visitor.visit_root_nan_atom(*this, value, TAPE_ERROR);
+    case 'i':
+    case 'I': return visitor.visit_root_inf_atom(*this, value);
+#else
     case 'n': return visitor.visit_root_null_atom(*this, value);
+#endif
     case '-':
     case '0': case '1': case '2': case '3': case '4':
     case '5': case '6': case '7': case '8': case '9':
@@ -47719,7 +59632,20 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::visit_primitive(V
   switch (*value) {
     case 't': return visitor.visit_true_atom(*this, value);
     case 'f': return visitor.visit_false_atom(*this, value);
+#if SIMDJSON_ENABLE_NAN_INF
+    case 'n': {
+      auto err = visitor.visit_null_atom(*this, value);
+      if (err == SUCCESS) { return err; }
+      // propagate the error value returned by a bad 'null' atom if parsing 'nan' fails
+      return visitor.visit_nan_atom(*this, value, err);
+    }
+    // 'N' isn't a canonically recognized atom, so we return a TAPE_ERROR if failure occurs
+    case 'N': return visitor.visit_nan_atom(*this, value, TAPE_ERROR);
+    case 'i':
+    case 'I': return visitor.visit_inf_atom(*this, value);
+#else
     case 'n': return visitor.visit_null_atom(*this, value);
+#endif
     default:
       log_error("Non-value found when value was expected!");
       return TAPE_ERROR;
@@ -47905,9 +59831,13 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_string(const uint8_t *src, u
            within the unicode codepoint handling code. */
         src += bs_dist;
         dst += bs_dist;
-        if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
-          return nullptr;
-        }
+        // Decode adjacent Unicode escapes without returning to the
+        // quote-and-backslash scanner between code points.
+        do {
+          if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
+            return nullptr;
+          }
+        } while (src[0] == '\\' && src[1] == 'u');
       } else {
         /* simple 1:1 conversion. Will eat bs_dist+2 characters in input and
          * write bs_dist+1 characters to output
@@ -47930,6 +59860,77 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_string(const uint8_t *src, u
   }
 }

+/**
+ * Bounds-safe variant of parse_string for input buffers that are NOT padded to
+ * len + SIMDJSON_PADDING bytes. `buf_end` is one past the last readable input
+ * byte (buf + len). It behaves exactly like parse_string while we are at least
+ * SIMDJSON_PADDING bytes away from buf_end (so every speculative SIMD read stays
+ * in bounds); once within the final SIMDJSON_PADDING bytes it copies the few
+ * remaining bytes into a space-padded scratch buffer and finishes with the
+ * regular parse_string. This keeps the delicate escape/Unicode handling in one
+ * place (the proven parse_string) rather than duplicating it.
+ *
+ * Correctness relies on stage 1 having validated the string, i.e. there is an
+ * unescaped closing quote within [src, buf_end); that quote is therefore inside
+ * the copied scratch, so parse_string finds it without running off the scratch.
+ */
+simdjson_warn_unused simdjson_inline uint8_t *parse_string_safe(const uint8_t *src, uint8_t *dst, bool allow_replacement, const uint8_t *buf_end) {
+  // Far from the end: identical to parse_string's loop. The guard uses
+  // SIMDJSON_PADDING (>= BYTES_PROCESSED) so copy_and_find never reads past
+  // buf_end; escape/Unicode look-aheads read within the string (before the
+  // closing quote, which is < buf_end), so they are in bounds here too.
+  // We add margin (+12) for handle_unicode_codepoint's worst-case lookahead:
+  // after seeing a high surrogate, it does hex_to_u32_nocheck on the immediate
+  // following bytes (+6 from the '\'), then (if it sees \u) another
+  // hex_to_u32_nocheck at +8..+11 relative to the backslash that started the
+  // escape. With bs_dist up to BYTES_PROCESSED-1 this reaches +11 from the
+  // chunk start. The +12 margin ensures that even on kernels where
+  // BYTES_PROCESSED == SIMDJSON_PADDING (e.g. icelake) the 4-byte read stays
+  // in-bounds. The scratch fallback (3*PAD) is already safe.
+  while (src + SIMDJSON_PADDING + 12 <= buf_end) {
+    auto b = backslash_and_quote{};
+    auto bs_quote = b.copy_and_find(src, dst);
+    if (bs_quote.has_quote_first()) {
+      return dst + bs_quote.quote_index();
+    }
+    if (bs_quote.has_backslash()) {
+      auto bs_dist = bs_quote.backslash_index();
+      uint8_t escape_char = src[bs_dist + 1];
+      if (escape_char == 'u') {
+        src += bs_dist;
+        dst += bs_dist;
+        if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
+          return nullptr;
+        }
+      } else {
+        uint8_t escape_result = escape_map[escape_char];
+        if (escape_result == 0u) {
+          return nullptr;
+        }
+        dst[bs_dist] = escape_result;
+        src += bs_dist + 2;
+        dst += bs_dist + 1;
+      }
+    } else {
+      src += backslash_and_quote::BYTES_PROCESSED;
+      dst += backslash_and_quote::BYTES_PROCESSED;
+    }
+  }
+  // Within the final SIMDJSON_PADDING bytes: copy what remains into a
+  // space-padded scratch (spaces are neither quote nor backslash, so they do not
+  // disturb matching) and let the regular parser finish from there. The closing
+  // quote is within `remaining` (< SIMDJSON_PADDING), so parse_string finds it in
+  // the chunk starting at some offset <= remaining and reads at most
+  // BYTES_PROCESSED (<= SIMDJSON_PADDING) further -- i.e. under 2*SIMDJSON_PADDING.
+  // We size at 3x for a comfortable margin (the unicode look-ahead reads a few
+  // extra bytes past an escape).
+  uint8_t scratch[SIMDJSON_PADDING * 3];
+  const size_t remaining = size_t(buf_end - src); // < SIMDJSON_PADDING
+  std::memset(scratch, ' ', sizeof(scratch));
+  std::memcpy(scratch, src, remaining);
+  return parse_string(scratch, dst, allow_replacement);
+}
+
 simdjson_warn_unused simdjson_inline uint8_t *parse_wobbly_string(const uint8_t *src, uint8_t *dst) {
   // It is not ideal that this function is nearly identical to parse_string.
   while (1) {
@@ -47984,73 +59985,6 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_wobbly_string(const uint8_t

 #endif // SIMDJSON_SRC_GENERIC_STAGE2_STRINGPARSING_H
 /* end file generic/stage2/stringparsing.h for lasx */
-/* including generic/stage2/structural_iterator.h for lasx: #include <generic/stage2/structural_iterator.h> */
-/* begin file generic/stage2/structural_iterator.h for lasx */
-#ifndef SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H
-
-/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
-/* amalgamation skipped (editor-only): #define SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H */
-/* amalgamation skipped (editor-only): #include <generic/stage2/base.h> */
-/* amalgamation skipped (editor-only): #include <simdjson/generic/dom_parser_implementation.h> */
-/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
-
-namespace simdjson {
-namespace lasx {
-namespace {
-namespace stage2 {
-
-class structural_iterator {
-public:
-  const uint8_t* const buf;
-  uint32_t *next_structural;
-  dom_parser_implementation &dom_parser;
-
-  // Start a structural
-  simdjson_inline structural_iterator(dom_parser_implementation &_dom_parser, size_t start_structural_index)
-    : buf{_dom_parser.buf},
-      next_structural{&_dom_parser.structural_indexes[start_structural_index]},
-      dom_parser{_dom_parser} {
-  }
-  // Get the buffer position of the current structural character
-  simdjson_inline const uint8_t* current() {
-    return &buf[*(next_structural-1)];
-  }
-  // Get the current structural character
-  simdjson_inline char current_char() {
-    return buf[*(next_structural-1)];
-  }
-  // Get the next structural character without advancing
-  simdjson_inline char peek_next_char() {
-    return buf[*next_structural];
-  }
-  simdjson_inline const uint8_t* peek() {
-    return &buf[*next_structural];
-  }
-  simdjson_inline const uint8_t* advance() {
-    return &buf[*(next_structural++)];
-  }
-  simdjson_inline char advance_char() {
-    return buf[*(next_structural++)];
-  }
-  simdjson_inline size_t remaining_len() {
-    return dom_parser.len - *(next_structural-1);
-  }
-
-  simdjson_inline bool at_end() {
-    return next_structural == &dom_parser.structural_indexes[dom_parser.n_structural_indexes];
-  }
-  simdjson_inline bool at_beginning() {
-    return next_structural == dom_parser.structural_indexes.get();
-  }
-};
-
-} // namespace stage2
-} // unnamed namespace
-} // namespace lasx
-} // namespace simdjson
-
-#endif // SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H
-/* end file generic/stage2/structural_iterator.h for lasx */
 /* including generic/stage2/tape_builder.h for lasx: #include <generic/stage2/tape_builder.h> */
 /* begin file generic/stage2/tape_builder.h for lasx */
 #ifndef SIMDJSON_SRC_GENERIC_STAGE2_TAPE_BUILDER_H
@@ -48073,12 +60007,8 @@ namespace lasx {
 namespace {
 namespace stage2 {

-struct tape_builder {
-  template<bool STREAMING>
-  simdjson_warn_unused static simdjson_inline error_code parse_document(
-    dom_parser_implementation &dom_parser,
-    dom::document &doc) noexcept;
-
+template <bool UNPADDED>
+struct tape_builder_impl {
   /** Called when a non-empty document starts. */
   simdjson_warn_unused simdjson_inline error_code visit_document_start(json_iterator &iter) noexcept;
   /** Called when a non-empty document ends without error. */
@@ -48131,88 +60061,130 @@ struct tape_builder {
   simdjson_warn_unused simdjson_inline error_code visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept;
   simdjson_warn_unused simdjson_inline error_code visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept;

+#if SIMDJSON_ENABLE_NAN_INF
+  simdjson_warn_unused simdjson_inline error_code visit_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept;
+  simdjson_warn_unused simdjson_inline error_code visit_root_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept;
+  // Attempts to parse 'inf' or 'infinity' (case insensitive). Because neither are canonical atoms,
+  // this returns a tape error on failure.
+  simdjson_warn_unused simdjson_inline error_code visit_inf_atom(json_iterator &iter, const uint8_t *value) noexcept;
+  simdjson_warn_unused simdjson_inline error_code visit_root_inf_atom(json_iterator &iter, const uint8_t *value) noexcept;
+#endif
+
   /** Called each time a new field or element in an array or object is found. */
   simdjson_warn_unused simdjson_inline error_code increment_count(json_iterator &iter) noexcept;

   /** Next location to write to tape */
   tape_writer tape;
+public:
+  simdjson_inline tape_builder_impl(dom::document &doc) noexcept;
 private:
   /** Next write location in the string buf for stage 2 parsing */
   uint8_t *current_string_buf_loc;

-  simdjson_inline tape_builder(dom::document &doc) noexcept;
-
   simdjson_inline uint32_t next_tape_index(json_iterator &iter) const noexcept;
   simdjson_inline void start_container(json_iterator &iter) noexcept;
   simdjson_warn_unused simdjson_inline error_code end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept;
   simdjson_warn_unused simdjson_inline error_code empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept;
   simdjson_inline uint8_t *on_start_string(json_iterator &iter) noexcept;
   simdjson_inline void on_end_string(uint8_t *dst) noexcept;
-}; // struct tape_builder
+}; // struct tape_builder_impl

-template<bool STREAMING>
-simdjson_warn_unused simdjson_inline error_code tape_builder::parse_document(
-    dom_parser_implementation &dom_parser,
-    dom::document &doc) noexcept {
-  dom_parser.doc = &doc;
-  json_iterator iter(dom_parser, STREAMING ? dom_parser.next_structural_index : 0);
-  tape_builder builder(doc);
-  return iter.walk_document<STREAMING>(builder);
-}
+// Thin, non-templated entry so each architecture's stage2() keeps calling
+// tape_builder::parse_document<STREAMING> unchanged. It chooses the bounds-safe
+// (unpadded) or the regular (padded) tape_builder_impl ONCE per document, so the
+// choice is a compile-time constant inside the walk: the padded path carries no
+// extra branch or load (see tape_builder_impl::visit_string).
+struct tape_builder {
+  template<bool STREAMING>
+  simdjson_warn_unused static simdjson_inline error_code parse_document(
+      dom_parser_implementation &dom_parser, dom::document &doc) noexcept {
+    dom_parser.doc = &doc;
+    json_iterator iter(dom_parser, STREAMING ? dom_parser.next_structural_index : 0);
+    if (dom_parser._unpadded) {
+      tape_builder_impl<true> builder(doc);
+      return iter.walk_document<STREAMING, true>(builder);
+    } else {
+      tape_builder_impl<false> builder(doc);
+      return iter.walk_document<STREAMING, false>(builder);
+    }
+  }
+};

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_primitive(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_primitive(json_iterator &iter, const uint8_t *value) noexcept {
   return iter.visit_root_primitive(*this, value);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_primitive(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_primitive(json_iterator &iter, const uint8_t *value) noexcept {
   return iter.visit_primitive(*this, value);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_empty_object(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_empty_object(json_iterator &iter) noexcept {
   return empty_container(iter, internal::tape_type::START_OBJECT, internal::tape_type::END_OBJECT);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_empty_array(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_empty_array(json_iterator &iter) noexcept {
   return empty_container(iter, internal::tape_type::START_ARRAY, internal::tape_type::END_ARRAY);
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_document_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_document_start(json_iterator &iter) noexcept {
   start_container(iter);
   return SUCCESS;
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_object_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_object_start(json_iterator &iter) noexcept {
   start_container(iter);
   return SUCCESS;
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_array_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_array_start(json_iterator &iter) noexcept {
   start_container(iter);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_object_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_object_end(json_iterator &iter) noexcept {
   return end_container(iter, internal::tape_type::START_OBJECT, internal::tape_type::END_OBJECT);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_array_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_array_end(json_iterator &iter) noexcept {
   return end_container(iter, internal::tape_type::START_ARRAY, internal::tape_type::END_ARRAY);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_document_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_document_end(json_iterator &iter) noexcept {
   constexpr uint32_t start_tape_index = 0;
   tape.append(start_tape_index, internal::tape_type::ROOT);
   tape_writer::write(iter.dom_parser.doc->tape[start_tape_index], next_tape_index(iter), internal::tape_type::ROOT);
   return SUCCESS;
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_key(json_iterator &iter, const uint8_t *key) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_key(json_iterator &iter, const uint8_t *key) noexcept {
   return visit_string(iter, key, true);
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::increment_count(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::increment_count(json_iterator &iter) noexcept {
   iter.dom_parser.open_containers[iter.depth].count++; // we have a key value pair in the object at parser.dom_parser.depth - 1
   return SUCCESS;
 }

-simdjson_inline tape_builder::tape_builder(dom::document &doc) noexcept : tape{doc.tape.get()}, current_string_buf_loc{doc.string_buf.get()} {}
+template <bool UNPADDED>
+simdjson_inline tape_builder_impl<UNPADDED>::tape_builder_impl(dom::document &doc) noexcept : tape{doc.tape.get()}, current_string_buf_loc{doc.string_buf.get()} {}

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_string(json_iterator &iter, const uint8_t *value, bool key) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_string(json_iterator &iter, const uint8_t *value, bool key) noexcept {
   iter.log_value(key ? "key" : "string");
   uint8_t *dst = on_start_string(iter);
-  dst = stringparsing::parse_string(value+1, dst, false); // We do not allow replacement when the escape characters are invalid.
+  // We do not allow replacement when the escape characters are invalid.
+  // UNPADDED is a compile-time constant chosen once per document by
+  // tape_builder::parse_document, so the padded build instantiates only the
+  // plain parse_string call below -- no runtime branch and no flag load.
+  SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+    dst = stringparsing::parse_string_safe(value+1, dst, false, iter.buf + iter.dom_parser.len);
+  } else {
+    dst = stringparsing::parse_string(value+1, dst, false);
+  }
   if (dst == nullptr) {
     iter.log_error("Invalid escape in string");
     return STRING_ERROR;
@@ -48221,27 +60193,48 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_string(json_
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_string(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_string(json_iterator &iter, const uint8_t *value) noexcept {
   return visit_string(iter, value);
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_number(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_number(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("number");
-  error_code err = numberparsing::parse_number(value, tape);
+  const uint8_t *num = value;
+  std::unique_ptr<uint8_t[]> copy{}; // keeps a padded copy of the tail alive when used
+  SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+    // numberparsing reads ahead in 8-byte blocks for floats
+    // (is_made_of_eight_digits_fast reads up to 7 bytes past the digits), so a
+    // number whose digits reach the final bytes of an unpadded buffer would read
+    // past it. *(next_structural) is the offset of the token following this
+    // number, hence an upper bound on where the digits end; when that is within
+    // SIMDJSON_PADDING of the end we parse from a space-padded copy of the tail
+    // (mirroring visit_root_number). This fires only for numbers near the end.
+    if (simdjson_unlikely(*(iter.next_structural) + SIMDJSON_PADDING > iter.dom_parser.len)) {
+      const size_t rl = iter.remaining_len(); // bytes from `value` to the end of the document
+      copy.reset(new (std::nothrow) uint8_t[rl + SIMDJSON_PADDING]);
+      if (copy.get() == nullptr) { return MEMALLOC; }
+      std::memcpy(copy.get(), value, rl);
+      std::memset(copy.get() + rl, ' ', SIMDJSON_PADDING);
+      num = copy.get();
+    }
+  }
+  error_code err = numberparsing::parse_number(num, tape);
   if (simdjson_unlikely(err == BIGINT_ERROR &&
       iter.dom_parser._number_as_string)) {
     // Write big integer to string buffer using the same format as strings.
     // Scan digits the same way parse_number does (skip optional '-', then digits).
-    const uint8_t *p = value;
+    const uint8_t *p = num;
     if (*p == '-') p++;
     while (numberparsing::is_digit(*p)) p++;
     // The digit run must be terminated by a structural or whitespace character; otherwise the
     // token is malformed (e.g. "123456789123456789123x").
     if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
-    size_t len = size_t(p - value);
+    size_t len = size_t(p - num);
     tape.append(current_string_buf_loc - iter.dom_parser.doc->string_buf.get(), internal::tape_type::BIGINT);
     uint8_t *dst = current_string_buf_loc + sizeof(uint32_t);
-    memcpy(dst, value, len);
+    memcpy(dst, num, len);
     dst += len;
     on_end_string(dst);
     return SUCCESS;
@@ -48249,7 +60242,8 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_number(json_
   return err;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_number(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_number(json_iterator &iter, const uint8_t *value) noexcept {
   //
   // We need to make a copy to make sure that the string is space terminated.
   // This is not about padding the input, which should already padded up
@@ -48263,76 +60257,142 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_number(
   // practice unless you are in the strange scenario where you have many JSON
   // documents made of single atoms.
   //
-  std::unique_ptr<uint8_t[]>copy(new (std::nothrow) uint8_t[iter.remaining_len() + SIMDJSON_PADDING]);
+  // In a stream, the input goes on with other documents: copy up to the next
+  // structural only, not to the end of the batch.
+  const size_t len = (std::min)(iter.remaining_len(), size_t(*iter.next_structural) - size_t(*(iter.next_structural - 1)));
+  std::unique_ptr<uint8_t[]>copy(new (std::nothrow) uint8_t[len + SIMDJSON_PADDING]);
   if (copy.get() == nullptr) { return MEMALLOC; }
-  std::memcpy(copy.get(), value, iter.remaining_len());
-  std::memset(copy.get() + iter.remaining_len(), ' ', SIMDJSON_PADDING);
+  std::memcpy(copy.get(), value, len);
+  std::memset(copy.get() + len, ' ', SIMDJSON_PADDING);
   error_code error = visit_number(iter, copy.get());
   return error;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("true");
-  if (!atomparsing::is_valid_true_atom(value)) { return T_ATOM_ERROR; }
+  // The non-length-aware validator reads a fixed 5 bytes; a malformed/truncated
+  // token at the very end of an unpadded buffer would over-read. Use the
+  // length-aware form there (the root variant already does this).
+  const bool ok = UNPADDED ? atomparsing::is_valid_true_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_true_atom(value);
+  if (!ok) { return T_ATOM_ERROR; }
   tape.append(0, internal::tape_type::TRUE_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("true");
   if (!atomparsing::is_valid_true_atom(value, iter.remaining_len())) { return T_ATOM_ERROR; }
   tape.append(0, internal::tape_type::TRUE_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("false");
-  if (!atomparsing::is_valid_false_atom(value)) { return F_ATOM_ERROR; }
+  const bool ok = UNPADDED ? atomparsing::is_valid_false_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_false_atom(value);
+  if (!ok) { return F_ATOM_ERROR; }
   tape.append(0, internal::tape_type::FALSE_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("false");
   if (!atomparsing::is_valid_false_atom(value, iter.remaining_len())) { return F_ATOM_ERROR; }
   tape.append(0, internal::tape_type::FALSE_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("null");
-  if (!atomparsing::is_valid_null_atom(value)) { return N_ATOM_ERROR; }
+  const bool ok = UNPADDED ? atomparsing::is_valid_null_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_null_atom(value);
+  if (!ok) { return N_ATOM_ERROR; }
   tape.append(0, internal::tape_type::NULL_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("null");
   if (!atomparsing::is_valid_null_atom(value, iter.remaining_len())) { return N_ATOM_ERROR; }
   tape.append(0, internal::tape_type::NULL_VALUE);
   return SUCCESS;
 }

+#if SIMDJSON_ENABLE_NAN_INF
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept {
+  iter.log_value("nan");
+  // For unpadded input use the length-aware validator so the 'infinity'-style
+  // 8-byte compare cannot read past the buffer on a malformed token at the end.
+  const bool ok = UNPADDED ? atomparsing::is_valid_nan_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_nan_atom(value);
+  if (!ok) { return errc; }
+  tape.append_double(std::numeric_limits<double>::quiet_NaN());
+  return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept {
+  iter.log_value("nan");
+  if (!atomparsing::is_valid_nan_atom(value, iter.remaining_len())) { return errc; }
+  tape.append_double(std::numeric_limits<double>::quiet_NaN());
+  return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_inf_atom(json_iterator &iter, const uint8_t *value) noexcept {
+  iter.log_value("inf");
+  // Because 'inf' is an extension, non a canonical atom, a tape error should be returned on failure.
+  // For unpadded input use the length-aware validator so the 'infinity' 8-byte
+  // compare cannot read past the buffer on a malformed token at the end.
+  const bool ok = UNPADDED ? atomparsing::is_valid_inf_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_inf_atom(value);
+  if (!ok) { return TAPE_ERROR; }
+  tape.append_double(std::numeric_limits<double>::infinity());
+  return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_inf_atom(json_iterator &iter, const uint8_t *value) noexcept {
+  iter.log_value("inf");
+  // Because 'inf' is an extension, non a canonical atom, a tape error should be returned on failure
+  if (!atomparsing::is_valid_inf_atom(value, iter.remaining_len())) { return TAPE_ERROR; }
+  tape.append_double(std::numeric_limits<double>::infinity());
+  return SUCCESS;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
 // private:

-simdjson_inline uint32_t tape_builder::next_tape_index(json_iterator &iter) const noexcept {
+template <bool UNPADDED>
+simdjson_inline uint32_t tape_builder_impl<UNPADDED>::next_tape_index(json_iterator &iter) const noexcept {
   return uint32_t(tape.next_tape_loc - iter.dom_parser.doc->tape.get());
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
   auto start_index = next_tape_index(iter);
   tape.append(start_index+2, start);
   tape.append(start_index, end);
   return SUCCESS;
 }

-simdjson_inline void tape_builder::start_container(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_inline void tape_builder_impl<UNPADDED>::start_container(json_iterator &iter) noexcept {
   iter.dom_parser.open_containers[iter.depth].tape_index = next_tape_index(iter);
   iter.dom_parser.open_containers[iter.depth].count = 0;
   tape.skip(); // We don't actually *write* the start element until the end.
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
   // Write the ending tape element, pointing at the start location
   const uint32_t start_tape_index = iter.dom_parser.open_containers[iter.depth].tape_index;
   tape.append(start_tape_index, end);
@@ -48345,13 +60405,15 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::end_container(json
   return SUCCESS;
 }

-simdjson_inline uint8_t *tape_builder::on_start_string(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_inline uint8_t *tape_builder_impl<UNPADDED>::on_start_string(json_iterator &iter) noexcept {
   // we advance the point, accounting for the fact that we have a NULL termination
   tape.append(current_string_buf_loc - iter.dom_parser.doc->string_buf.get(), internal::tape_type::STRING);
   return current_string_buf_loc + sizeof(uint32_t);
 }

-simdjson_inline void tape_builder::on_end_string(uint8_t *dst) noexcept {
+template <bool UNPADDED>
+simdjson_inline void tape_builder_impl<UNPADDED>::on_end_string(uint8_t *dst) noexcept {
   uint32_t str_length = uint32_t(dst - (current_string_buf_loc + sizeof(uint32_t)));
   // TODO check for overflow in case someone has a crazy string (>=4GB?)
   // But only add the overflow check when the document itself exceeds 4GB
@@ -48613,10 +60675,6 @@ simdjson_inline int count_ones(uint64_t input_num) {
   return __lsx_vpickve2gr_w(__lsx_vpcnt_d(__m128i(v2u64{input_num, 0})), 0);
 }

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2, uint64_t *result) {
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-}

 } // unnamed namespace
 } // namespace lsx
@@ -49345,6 +61403,9 @@ namespace atomparsing {
 // to the compile-time constant 1936482662.
 simdjson_inline uint32_t string_to_uint32(const char* str) { uint32_t val; std::memcpy(&val, str, sizeof(uint32_t)); return val; }

+// Acts on the same principle as string_to_uint32, but on an 8-byte block of memory
+simdjson_inline uint64_t string_to_uint64(const char* str) { uint64_t val; std::memcpy(&val, str, sizeof(uint64_t)); return val; }
+

 // Again in str4ncmp we use a memcpy to avoid undefined behavior. The memcpy may appear expensive.
 // Yet all decent optimizing compilers will compile memcpy to a single instruction, just about.
@@ -49356,6 +61417,28 @@ simdjson_inline uint32_t str4ncmp(const uint8_t *src, const char* atom) {
   return srcval ^ string_to_uint32(atom);
 }

+// Checks that the first 8 characters of the input string match the given atom in a case-insensitive manner.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint64_t str8ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+  uint64_t srcval; // we want to avoid unaligned 32-bit loads (undefined in C/C++)
+  static_assert(sizeof(uint64_t) <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be larger than 8 bytes");
+  std::memcpy(&srcval, src, sizeof(uint64_t));
+
+  return (srcval | 0x2020202020202020ull) ^ string_to_uint64(atom);
+}
+
+// Checks that the first 3 characters of 'src' match 'atom' in a case-insensitive way.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint32_t str3ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+  return ((src[0] | 0x20) ^ atom[0]) //
+       | ((src[1] | 0x20) ^ atom[1]) //
+       | ((src[2] | 0x20) ^ atom[2]);
+}
+
 simdjson_warn_unused
 simdjson_inline bool is_valid_true_atom(const uint8_t *src) {
   return (str4ncmp(src, "true") | jsoncharutils::is_not_structural_or_whitespace(src[4])) == 0;
@@ -49392,6 +61475,71 @@ simdjson_inline bool is_valid_null_atom(const uint8_t *src, size_t len) {
   else { return false; }
 }

+#if SIMDJSON_ENABLE_NAN_INF
+// "nan" is 3 bytes; we check characters and then verify the next
+// character is structural or whitespace. We accept both "nan" and "NaN".
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src) {
+  return (str3ncmp_case_insensitive(src, "nan")
+        | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+}
+
+// checks that the next four characters of a string are 'nan"', where the 'nan'
+// is checked in a case-insensitive way.
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_in_string(const uint8_t *src) {
+  return (str3ncmp_case_insensitive(src, "nan") | (src[3] ^ '"')) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src, size_t len) {
+  if (len > 3) { return is_valid_nan_atom(src); }
+  if (len == 3) { return str3ncmp_case_insensitive(src, "nan") == 0; }
+  return false;
+}
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src) {
+  bool is_short_inf = (str3ncmp_case_insensitive(src, "inf")
+                     | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+  if(is_short_inf) return true;
+
+  // Check for 'infinity' (any capitalization)
+  return (str8ncmp_case_insensitive(src, "infinity") | jsoncharutils::is_not_structural_or_whitespace(src[8])) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_in_string(const uint8_t *src) {
+  bool is_short_inf = (str3ncmp_case_insensitive(src, "inf") | (src[3] ^ '"')) == 0;
+  if(is_short_inf) return true;
+
+  return (str8ncmp_case_insensitive(src, "infinity") | (src[8] ^ '"')) == 0;
+}
+
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src, size_t len) {
+  if (len > 8) { return is_valid_inf_atom(src); }
+  if (len == 8 && str8ncmp_case_insensitive(src, "infinity") == 0) {
+    return true;
+  }
+  if (len > 3) {
+    return (str3ncmp_case_insensitive(src, "inf")
+          | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+  }
+  if (len == 3) { return str3ncmp_case_insensitive(src, "inf") == 0; }
+  return false;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
 } // namespace atomparsing
 } // unnamed namespace
 } // namespace lsx
@@ -49482,7 +61630,7 @@ inline simdjson_warn_unused error_code dom_parser_implementation::set_capacity(s
 }

 inline simdjson_warn_unused error_code dom_parser_implementation::set_max_depth(size_t max_depth) noexcept {
-  if(max_depth == 0 || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
+  if(max_depth == 0 || max_depth > SIMDJSON_MAX_DEPTH || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
   // Stage 2 stacks
   open_containers.reset(new (std::nothrow) open_container[max_depth]);
   is_array.reset(new (std::nothrow) bool[max_depth]);
@@ -49661,6 +61809,7 @@ protected:
 /* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_NUMBERPARSING_H */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/base.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/jsoncharutils.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/atomparsing.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/internal/numberparsing_tables.h" */
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */

@@ -49700,6 +61849,18 @@ simdjson_inline double to_double(uint64_t mantissa, uint64_t real_exponent, bool
     return d;
 }

+// Convert a mantissa, an exponent and a sign bit into an ieee32 float (binary32).
+// The real_exponent needs to be in [0, 254] (technically real_exponent = 255 would be acceptable).
+// The mantissa should be in [0,1<<24). The bit at index (1U << 23) will be zeroed.
+simdjson_inline float to_float(uint32_t mantissa, uint32_t real_exponent, bool negative) {
+    float f;
+    mantissa &= ~(uint32_t(1) << 23);
+    mantissa |= real_exponent << 23;
+    mantissa |= ((static_cast<uint32_t>(negative)) << 31);
+    std::memcpy(&f, &mantissa, sizeof(f));
+    return f;
+}
+
 // Attempts to compute i * 10^(power) exactly; and if "negative" is
 // true, negate the result.
 // This function will only work in some cases, when it does not work, success is
@@ -49956,6 +62117,219 @@ simdjson_inline bool compute_float_64(int64_t power, uint64_t i, bool negative,
   return true;
 }

+// Attempts to compute i * 10^(power) as a binary32 (float) value; and if
+// "negative" is true, negate the result.
+//
+// This is the same Eisel-Lemire algorithm as compute_float_64, retargeted at
+// binary32. It is adapted from fast_float
+// (https://github.com/fastfloat/fast_float), where the routine is written once
+// and instantiated for each binary format. We only need 24 bits of mantissa
+// instead of 53, so we ask the truncated product for 24 + 2 = 26 accurate bits
+// (plus one bit that may be lost to the "upperbit" shift), and we shift the
+// product right by (upperbit + 64 - 23 - 3) instead of (upperbit + 64 - 52 - 3).
+//
+// The power-of-five table is shared with the binary64 code: the useful range
+// for binary32, [smallest_power_binary32, largest_power_binary32], is a strict
+// subset of [smallest_power, largest_power].
+//
+// The function returns false when the result would be infinite: simdjson
+// refuses to parse infinite values, so the caller reports an error. Unlike
+// compute_float_64, a false return never means "try harder": the accuracy of
+// the two-limb product is guaranteed by Noble Mushtak and Daniel Lemire,
+// "Fast Number Parsing Without Fallback" (https://arxiv.org/abs/2212.06644).
+// The caller must still fall back to a slow path when the 64-bit mantissa i
+// was truncated (more than 19 significant digits).
+//
+// We assume that power is in the [smallest_power, largest_power] interval: the
+// caller is responsible for this check.
+simdjson_inline bool compute_float_32(int64_t power, uint64_t i, bool negative, float &d) {
+  // Powers of ten that are exactly representable as binary32 values: 10^k is
+  // exact as long as k <= 10 (5^10 = 9765625 < 2**24 while 5^11 > 2**24).
+  static constexpr float power_of_ten_float[] = {
+      1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+      1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+  // The range of powers of ten that a non-zero, finite binary32 value can be
+  // built from. Anything smaller rounds to zero, anything larger is infinite.
+  // These are the smallest_power_of_ten()/largest_power_of_ten() constants that
+  // fast_float uses for binary32.
+  constexpr int smallest_power_binary32 = -65;
+  constexpr int largest_power_binary32 = 38;
+
+  // We start with the fast path described in
+  // Clinger WD. How to read floating point numbers accurately.
+  // ACM SIGPLAN Notices. 1990
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+  // We cannot be certain that x/y is rounded to nearest.
+  if (0 <= power && power <= 10 && i <= 16777215)
+#else
+  if (-10 <= power && power <= 10 && i <= 16777215)
+#endif
+  {
+    // Convert the integer into a float. This is lossless since
+    // 0 <= i <= 2^24 - 1.
+    d = float(i);
+    // Both d and the power of ten are exactly representable as binary32
+    // values, so the product (or quotient) is correctly rounded.
+    if (power < 0) {
+      d = d / power_of_ten_float[-power];
+    } else {
+      d = d * power_of_ten_float[power];
+    }
+    if (negative) {
+      d = -d;
+    }
+    return true;
+  }
+
+  // The fast path has failed, so we fall back on the Eisel-Lemire algorithm.
+  // It needs i > 0 (so that the leading bit of i can be normalized), so we
+  // handle i == 0 separately. We also handle the powers of ten that are so
+  // small (or so large) that the answer is zero (or infinite) whatever the
+  // mantissa is.
+  if (i == 0 || power < smallest_power_binary32) {
+    d = negative ? -0.0f : 0.0f;
+    return true;
+  }
+  if (power > largest_power_binary32) {
+    // We have, for sure, an infinite value.
+    return false;
+  }
+
+  // The exponent is 128 + 63 + power + floor(log(5**power)/log(2)).
+  // The 128 comes from the ieee32 standard (the minimal exponent is -127).
+  // The 63 comes from the fact that we use a 64-bit word.
+  // See compute_float_64 for a discussion of the magical 152170 + 65536.
+  int64_t exponent = (((152170 + 65536) * power) >> 16) + 128 + 63;
+
+  // We want the most significant bit of i to be 1. Shift if needed.
+  int lz = leading_zeroes(i);
+  i <<= lz;
+
+  // We want the most significant 64 bits of the product i * 5**power. It is
+  // safe to index the table because
+  // smallest_power <= smallest_power_binary32 <= power
+  //               <= largest_power_binary32 <= largest_power.
+  const uint32_t index = 2 * uint32_t(power - simdjson::internal::smallest_power);
+#if SIMDJSON_STATIC_REFLECTION
+  simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index]);
+#else
+  simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index]);
+#endif
+
+  // Unless the least significant 38 bits of the high (64-bit) part of the full
+  // product are all 1s, then we know that the most significant 26 bits are
+  // exact and no further work is needed. Having 26 bits is necessary because
+  // we need 24 bits for the mantissa but we have to have one rounding bit and
+  // we can waste a bit if the most significant bit of the product is zero.
+  // (For binary64, the same reasoning gives 55 bits and a 9-bit mask.)
+  if((firstproduct.high & 0x3FFFFFFFFF) == 0x3FFFFFFFFF) {
+    // The truncated multiplication was not accurate enough; use the next 64
+    // bits of the power of five to refine it. See compute_float_64 for a
+    // detailed discussion.
+#if SIMDJSON_STATIC_REFLECTION
+    simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index + 1]);
+#else
+    simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index + 1]);
+#endif
+    firstproduct.low += secondproduct.high;
+    if(secondproduct.high > firstproduct.low) { firstproduct.high++; }
+  }
+  uint64_t lower = firstproduct.low;
+  uint64_t upper = firstproduct.high;
+  // The final mantissa should be 24 bits with a leading 1.
+  // We shift it so that it occupies 25 bits with a leading 1.
+  ///////
+  uint64_t upperbit = upper >> 63;
+  uint64_t mantissa = upper >> (upperbit + 38); // 38 == 64 - 23 - 3
+  lz += int(1 ^ upperbit);
+
+  // Here we have mantissa < (1<<25).
+  int64_t real_exponent = exponent - lz;
+  if (simdjson_unlikely(real_exponent <= 0)) { // we have a subnormal?
+    // Here we have that real_exponent <= 0 so -real_exponent >= 0
+    if(-real_exponent + 1 >= 64) { // if we have more than 64 bits below the minimum exponent, you have a zero for sure.
+      d = negative ? -0.0f : 0.0f;
+      return true;
+    }
+    // next line is safe because -real_exponent + 1 < 64
+    mantissa >>= -real_exponent + 1;
+    // Thankfully, we can't have both "round-to-even" and subnormals because
+    // "round-to-even" only occurs for powers close to 0.
+    mantissa += (mantissa & 1); // round up
+    mantissa >>= 1;
+    // As in compute_float_64, rounding up may take us out of the subnormal
+    // range, so we can only decide after rounding.
+    real_exponent = (mantissa < (uint64_t(1) << 23)) ? 0 : 1;
+    d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+    return true;
+  }
+  // We have to round to even. The "to even" part is only a problem when we are
+  // right in between two floats, which we guard against. The bounds on the
+  // power of ten are those of fast_float's
+  // min_exponent_round_to_even()/max_exponent_round_to_even() for binary32:
+  // when q >= 0, (2m+1) must be divisible by 5^q with 5^q <= 2^25, so q <= 10;
+  // when q < 0, we need 2^24 x 5^{-q} < 2^{64}, so q >= -17.
+  if (simdjson_unlikely((lower <= 1) && (power >= -17) && (power <= 10) && ((mantissa & 3) == 1))) {
+    if((mantissa << (upperbit + 38)) == upper) {
+      mantissa &= ~uint64_t(1);   // flip it so that we do not round up
+    }
+  }
+
+  mantissa += mantissa & 1;
+  mantissa >>= 1;
+
+  // Here we have mantissa < (1<<24), unless there was an overflow
+  if (mantissa >= (uint64_t(1) << 24)) {
+    mantissa = (uint64_t(1) << 23);
+    real_exponent++;
+  }
+  mantissa &= ~(uint64_t(1) << 23);
+  // we have to check that real_exponent is in range, otherwise we bail out
+  if (simdjson_unlikely(real_exponent > 254)) {
+    // We have an infinite value!!! We could actually throw an error here if we could.
+    return false;
+  }
+  d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+  return true;
+}
+
+#if SIMDJSON_ENABLE_NAN_INF
+// Parses a nan or infinity. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, double& d) noexcept {
+  if (atomparsing::is_valid_inf_atom(src)) {
+    double inf = std::numeric_limits<double>::infinity();
+    d = negative ? -inf : inf;
+    return true;
+  }
+
+  if (atomparsing::is_valid_nan_atom(src)) {
+    d = std::numeric_limits<double>::quiet_NaN();
+    return true;
+  }
+
+  return false;
+}
+
+// Parses a nan or infinity as a binary32 value. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, float& d) noexcept {
+  if (atomparsing::is_valid_inf_atom(src)) {
+    float inf = std::numeric_limits<float>::infinity();
+    d = negative ? -inf : inf;
+    return true;
+  }
+
+  if (atomparsing::is_valid_nan_atom(src)) {
+    d = std::numeric_limits<float>::quiet_NaN();
+    return true;
+  }
+
+  return false;
+}
+#endif
+
 // We call a fallback floating-point parser that might be slow. Note
 // it will accept JSON numbers, but the JSON spec. is more restrictive so
 // before you call parse_float_fallback, you need to have validated the input
@@ -49993,6 +62367,16 @@ static bool parse_float_fallback(const uint8_t *ptr, const uint8_t *end_ptr, dou
   return !(*outDouble > (std::numeric_limits<double>::max)() || *outDouble < std::numeric_limits<double>::lowest());
 }

+// Same as parse_float_fallback, but for binary32 (float) values. Going through
+// the binary64 fallback and then rounding to binary32 would be subject to
+// double rounding, so we run the fallback algorithm directly on binary32.
+static bool parse_float_fallback(const uint8_t *ptr, float *outFloat) {
+  *outFloat = simdjson::internal::from_chars_float(reinterpret_cast<const char *>(ptr));
+  // We do not accept infinite values. See the binary64 version above for why we
+  // do not use std::isfinite.
+  return !(*outFloat > (std::numeric_limits<float>::max)() || *outFloat < std::numeric_limits<float>::lowest());
+}
+
 // check quickly whether the next 8 chars are made of digits
 // at a glance, it looks better than Mula's
 // http://0x80.pl/articles/swar-digits-validate.html
@@ -50011,6 +62395,28 @@ simdjson_inline bool is_made_of_eight_digits_fast(const uint8_t *chars) {
           0x3333333333333333);
 }

+// The same idea for four characters. A block of eight only bites when eight
+// digits are left, so a run of 4 to 7 digits used to fall back to one digit at a
+// time; taking four of them at once is what makes the tail of a long fraction
+// cheap. Adapted from fast_float, credit @aqrit.
+simdjson_inline bool is_made_of_four_digits_fast(const uint8_t *chars) {
+  uint32_t val;
+  // Reads up to 3 bytes beyond the digits, which SIMDJSON_PADDING covers.
+  static_assert(3 <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be bigger than 3");
+  std::memcpy(&val, chars, 4);
+  return !(((val + 0x46464646) | (val - 0x30303030)) & 0x80808080);
+}
+
+// Only call this when is_made_of_four_digits_fast() says the four characters are
+// digits. Little-endian: guarded by SIMDJSON_SWAR_NUMBER_PARSING.
+simdjson_inline uint32_t parse_four_digits_unrolled(const uint8_t *chars) {
+  uint32_t val;
+  std::memcpy(&val, chars, 4);
+  val -= 0x30303030;
+  val = (val * 10) + (val >> 8);
+  return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
 template<typename I>
 SIMDJSON_NO_SANITIZE_UNDEFINED // We deliberately allow overflow here and check later
 simdjson_inline bool parse_digit(const uint8_t c, I &i) {
@@ -50027,26 +62433,73 @@ simdjson_inline bool is_digit(const uint8_t c) {
   return static_cast<uint8_t>(c - '0') <= 9;
 }

-simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
-  // we continue with the fiction that we have an integer. If the
-  // floating point number is representable as x * 10^z for some integer
-  // z that fits in 53 bits, then we will be able to convert back the
-  // the integer into a float in a lossless manner.
-  const uint8_t *const first_after_period = p;
+// Consumes a run of digits into i, eight at a time while eight are available,
+// then four, then one at a time. Overflow is deliberate: the caller counts the
+// digits and falls back when there are too many for a 64-bit mantissa.
+//
+// Long runs of digits are what a float-heavy document is made of: the fraction
+// of a binary64 printed to full precision is 15 to 17 digits, and reading those
+// one at a time is the single largest cost in parse_double(). Integer parts are
+// left alone: they are usually a handful of digits, and there the check for a
+// block of eight is wasted work.
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_fraction_digits(const uint8_t *&p, uint64_t &i) {
+#ifdef SIMDJSON_SWAR_NUMBER_PARSING
+#if SIMDJSON_SWAR_NUMBER_PARSING
+  while (is_made_of_eight_digits_fast(p)) {
+    i = i * 100000000 + parse_eight_digits_unrolled(p);
+    p += 8;
+  }
+  // A 4 to 7 digit remainder is the common case once the blocks of eight are
+  // gone: 15 fraction digits leave 7, and 17 leave 1 after two blocks.
+  if (is_made_of_four_digits_fast(p)) {
+    i = i * 10000 + parse_four_digits_unrolled(p);
+    p += 4;
+  }
+#endif // SIMDJSON_SWAR_NUMBER_PARSING
+#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
+  while (parse_digit(*p, i)) { p++; }
+}

+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_float_integer_digits(const uint8_t *&p, uint64_t &i,
+                                                bool &leading_zero) {
+  if (!parse_digit(*p, i)) { leading_zero = true; return; }
+  p++;
+  leading_zero = (i == 0);
+  while (parse_digit(*p, i)) { p++; }
+}
+
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_integer_digits(const uint8_t *&p, uint64_t &i) {
 #ifdef SIMDJSON_SWAR_NUMBER_PARSING
 #if SIMDJSON_SWAR_NUMBER_PARSING
-  // this helps if we have lots of decimals!
-  // this turns out to be frequent enough.
+  // Identifiers, timestamps and counters often have eight digits or more.
+  // Prior related work: jsonifier parses integers as eight-digit SWAR words
+  // (str_to_i.hpp, https://github.com/nihilai-collective/Jsonifier). This
+  // takes one such word with parse_eight_digits_unrolled, the routine
+  // simdjson uses for long fractions.
   if (is_made_of_eight_digits_fast(p)) {
     i = i * 100000000 + parse_eight_digits_unrolled(p);
     p += 8;
   }
+  const uint8_t *const swar_end = p + 8;
+  while (p < swar_end && is_made_of_four_digits_fast(p)) {
+    i = i * 10000 + parse_four_digits_unrolled(p);
+    p += 4;
+  }
 #endif // SIMDJSON_SWAR_NUMBER_PARSING
 #endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
-  // Unrolling the first digit makes a small difference on some implementations (e.g. westmere)
-  if (parse_digit(*p, i)) { ++p; }
   while (parse_digit(*p, i)) { p++; }
+}
+
+simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
+  // we continue with the fiction that we have an integer. If the
+  // floating point number is representable as x * 10^z for some integer
+  // z that fits in 53 bits, then we will be able to convert back the
+  // the integer into a float in a lossless manner.
+  const uint8_t *const first_after_period = p;
+  parse_fraction_digits(p, i);
   exponent = first_after_period - p;
   // Decimal without digits (123.) is illegal
   if (exponent == 0) {
@@ -50135,7 +62588,7 @@ simdjson_inline size_t significant_digits(const uint8_t * start_digits, size_t d
 } // unnamed namespace

 /** @private */
-static error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
+inline error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
   if (parse_float_fallback(src, answer)) {
     return SUCCESS;
   }
@@ -50218,15 +62671,17 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
   return SUCCESS;              // always succeeds
 }

-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept  { return false; }
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept  { return false; }
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept { return number_type::signed_integer; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline bool is_negative(const uint8_t *) noexcept  { return false; }
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t *) noexcept  { return false; }
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t *) noexcept { return number_type::signed_integer; }
 #else

 // parse the number at src
@@ -50257,7 +62712,21 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
   size_t digit_count = size_t(p - start_digits);
-  if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) { return INVALID_NUMBER(src); }
+  if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // By this point, we know that our input does not begin with a digit. We will attempt
+    // to handle NaN/Infinity.
+
+    double d;
+    if (compute_nan_inf(p, negative, d)) {
+      writer.append_double(d);
+      return SUCCESS;
+    }
+#endif
+
+    return INVALID_NUMBER(src);
+  }

   //
   // Handle floats if there is a . or e (or both)
@@ -50306,7 +62775,7 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
     // - Therefore, if the number is positive and lower than that, it's overflow.
     // - The value we are looking at is less than or equal to INT64_MAX.
     //
-    }  else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return INVALID_NUMBER(src); }
+    }  else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return BIGINT_NUMBER(src); }
   }

   // Write unsigned if it does not fit in a signed integer.
@@ -50405,7 +62874,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const u
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -50503,7 +62972,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_stri
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -50558,7 +63027,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uin
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -50644,7 +63113,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = src;
   uint64_t i = 0;
-  while (parse_digit(*src, i)) { src++; }
+  parse_integer_digits(src, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -50684,9 +63153,247 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   //
   uint64_t i = 0;
   const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no loading digits, the number may be nan or infinity.
+    // Attempt to compute those, and return on success.
+    double d;
+    if (compute_nan_inf(p, negative, d)) { return d; }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = p-start_digits > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if (*p == 'e' || *p == 'E') {
+    p++;
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while (parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  double d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_64(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
+// Parse a JSON number into a binary32 (float) value.
+//
+// This mirrors parse_double, but it rounds to binary32 directly instead of
+// rounding to binary64 and then to binary32: the latter is subject to double
+// rounding and would not always produce the nearest float.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * src) noexcept {
+  //
+  // Check for minus sign
+  //
+  bool negative = (*src == '-');
+  src += uint8_t(negative);
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, the number may be nan or infinity.
+    // Attempt to compute those, and return on success.
+    float f;
+    if (compute_nan_inf(p, negative, f)) { return f; }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = p-start_digits > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if (*p == 'e' || *p == 'E') {
+    p++;
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while (parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  float d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_32(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
+simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
+  return (*src == '-');
+}
+
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept {
+  bool negative = (*src == '-');
+  src += uint8_t(negative);
+  const uint8_t *p = src;
+  while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
+  if ( p == src ) { return NUMBER_ERROR; }
+  if (jsoncharutils::is_structural_or_whitespace(*p)) { return true; }
+  return false;
+}
+
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept {
+  bool negative = (*src == '-');
+  src += uint8_t(negative);
+  const uint8_t *p = src;
+  while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
+  size_t digit_count = size_t(p - src);
+  if ( p == src ) { return NUMBER_ERROR; }
+  if (jsoncharutils::is_structural_or_whitespace(*p)) {
+    static const uint8_t * smaller_big_integer = reinterpret_cast<const uint8_t *>("9223372036854775808");
+    // We have an integer.
+    if(simdjson_unlikely(digit_count > 20)) {
+      return number_type::big_integer;
+    }
+    // If the number is negative and valid, it must be a signed integer.
+    if(negative) {
+      if (simdjson_unlikely(digit_count > 19)) return number_type::big_integer;
+      if (simdjson_unlikely(digit_count == 19 && memcmp(src, smaller_big_integer, 19) > 0)) {
+        return number_type::big_integer;
+      }
+#if SIMDJSON_MINUS_ZERO_AS_FLOAT
+      if(digit_count == 1 && src[0] == '0') {
+        // We have to write -0.0 instead of 0
+        return number_type::floating_point_number;
+      }
+#endif
+      return number_type::signed_integer;
+    }
+    // Let us check if we have a big integer (>=2**64).
+    static const uint8_t * two_to_sixtyfour = reinterpret_cast<const uint8_t *>("18446744073709551616");
+    if((digit_count > 20) || (digit_count == 20 && memcmp(src, two_to_sixtyfour, 20) >= 0)) {
+      return number_type::big_integer;
+    }
+    // The number is positive and smaller than 18446744073709551616 (or 2**64).
+    // We want values larger or equal to 9223372036854775808 to be unsigned
+    // integers, and the other values to be signed integers.
+    if((digit_count == 20) || (digit_count >= 19 && memcmp(src, smaller_big_integer, 19) >= 0)) {
+      return number_type::unsigned_integer;
+    }
+    return number_type::signed_integer;
+  }
+  // Hopefully, we have 'e' or 'E' or '.'.
+  return number_type::floating_point_number;
+}
+
+// Never read at src_end or beyond
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * src, const uint8_t * const src_end) noexcept {
+  if(src == src_end) { return NUMBER_ERROR; }
+  //
+  // Check for minus sign
+  //
+  bool negative = (*src == '-');
+  src += uint8_t(negative);
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  if(p == src_end) { return NUMBER_ERROR; }
   p += parse_digit(*p, i);
   bool leading_zero = (i == 0);
-  while (parse_digit(*p, i)) { p++; }
+  while ((p != src_end) && parse_digit(*p, i)) { p++; }
   // no integer digits, or 0123 (zero must be solo)
   if ( p == src ) { return INCORRECT_TYPE; }
   if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
@@ -50696,12 +63403,104 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   //
   int64_t exponent = 0;
   bool overflow;
-  if (simdjson_likely(*p == '.')) {
+  if (simdjson_likely((p != src_end) && (*p == '.'))) {
     p++;
     const uint8_t *start_decimal_digits = p;
-    if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
+    if ((p == src_end) || !parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
     p++;
-    while (parse_digit(*p, i)) { p++; }
+    while ((p != src_end) && parse_digit(*p, i)) { p++; }
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = start_digits-src > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if ((p != src_end) && (*p == 'e' || *p == 'E')) {
+    p++;
+    if(p == src_end) { return NUMBER_ERROR; }
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while ((p != src_end) && parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if ((p != src_end) && jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  double d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_64(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), src_end, &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * src) noexcept {
+  //
+  // Check for minus sign
+  //
+  bool negative = (*(src + 1) == '-');
+  src += uint8_t(negative) + 1;
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, attempt to parse numbers that are either
+    // NaN or Infinity
+    if (atomparsing::is_valid_inf_in_string(src)) {
+      double inf = std::numeric_limits<double>::infinity();
+      return negative ? -inf : inf;
+    }
+
+    if (atomparsing::is_valid_nan_in_string(src)) {
+      return std::numeric_limits<double>::quiet_NaN();
+    }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
     exponent = -(p - start_decimal_digits);

     // Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -50733,7 +63532,7 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
     exponent += exp_neg ? 0-exp : exp;
   }

-  if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+  if (*p != '"') { return NUMBER_ERROR; }

   overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;

@@ -50750,163 +63549,39 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   return d;
 }

-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
-  return (*src == '-');
-}
-
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept {
-  bool negative = (*src == '-');
-  src += uint8_t(negative);
-  const uint8_t *p = src;
-  while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
-  if ( p == src ) { return NUMBER_ERROR; }
-  if (jsoncharutils::is_structural_or_whitespace(*p)) { return true; }
-  return false;
-}
-
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept {
-  bool negative = (*src == '-');
-  src += uint8_t(negative);
-  const uint8_t *p = src;
-  while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
-  size_t digit_count = size_t(p - src);
-  if ( p == src ) { return NUMBER_ERROR; }
-  if (jsoncharutils::is_structural_or_whitespace(*p)) {
-    static const uint8_t * smaller_big_integer = reinterpret_cast<const uint8_t *>("9223372036854775808");
-    // We have an integer.
-    if(simdjson_unlikely(digit_count > 20)) {
-      return number_type::big_integer;
-    }
-    // If the number is negative and valid, it must be a signed integer.
-    if(negative) {
-      if (simdjson_unlikely(digit_count > 19)) return number_type::big_integer;
-      if (simdjson_unlikely(digit_count == 19 && memcmp(src, smaller_big_integer, 19) > 0)) {
-        return number_type::big_integer;
-      }
-#if SIMDJSON_MINUS_ZERO_AS_FLOAT
-      if(digit_count == 1 && src[0] == '0') {
-        // We have to write -0.0 instead of 0
-        return number_type::floating_point_number;
-      }
-#endif
-      return number_type::signed_integer;
-    }
-    // Let us check if we have a big integer (>=2**64).
-    static const uint8_t * two_to_sixtyfour = reinterpret_cast<const uint8_t *>("18446744073709551616");
-    if((digit_count > 20) || (digit_count == 20 && memcmp(src, two_to_sixtyfour, 20) >= 0)) {
-      return number_type::big_integer;
-    }
-    // The number is positive and smaller than 18446744073709551616 (or 2**64).
-    // We want values larger or equal to 9223372036854775808 to be unsigned
-    // integers, and the other values to be signed integers.
-    if((digit_count == 20) || (digit_count >= 19 && memcmp(src, smaller_big_integer, 19) >= 0)) {
-      return number_type::unsigned_integer;
-    }
-    return number_type::signed_integer;
-  }
-  // Hopefully, we have 'e' or 'E' or '.'.
-  return number_type::floating_point_number;
-}
-
-// Never read at src_end or beyond
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * src, const uint8_t * const src_end) noexcept {
-  if(src == src_end) { return NUMBER_ERROR; }
+// Parse a JSON number held inside a JSON string into a binary32 (float) value.
+// See parse_float for why we do not simply round parse_double_in_string.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * src) noexcept {
   //
   // Check for minus sign
   //
-  bool negative = (*src == '-');
-  src += uint8_t(negative);
+  bool negative = (*(src + 1) == '-');
+  src += uint8_t(negative) + 1;

   //
   // Parse the integer part.
   //
   uint64_t i = 0;
   const uint8_t *p = src;
-  if(p == src_end) { return NUMBER_ERROR; }
-  p += parse_digit(*p, i);
-  bool leading_zero = (i == 0);
-  while ((p != src_end) && parse_digit(*p, i)) { p++; }
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
   // no integer digits, or 0123 (zero must be solo)
-  if ( p == src ) { return INCORRECT_TYPE; }
-  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
-
-  //
-  // Parse the decimal part.
-  //
-  int64_t exponent = 0;
-  bool overflow;
-  if (simdjson_likely((p != src_end) && (*p == '.'))) {
-    p++;
-    const uint8_t *start_decimal_digits = p;
-    if ((p == src_end) || !parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
-    p++;
-    while ((p != src_end) && parse_digit(*p, i)) { p++; }
-    exponent = -(p - start_decimal_digits);
-
-    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
-    overflow = p-src-1 > 19;
-    if (simdjson_unlikely(overflow && leading_zero)) {
-      // Skip leading 0.00000 and see if it still overflows
-      const uint8_t *start_digits = src + 2;
-      while (*start_digits == '0') { start_digits++; }
-      overflow = start_digits-src > 19;
+  if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, attempt to parse numbers that are either
+    // NaN or Infinity
+    if (atomparsing::is_valid_inf_in_string(src)) {
+      float inf = std::numeric_limits<float>::infinity();
+      return negative ? -inf : inf;
     }
-  } else {
-    overflow = p-src > 19;
-  }
-
-  //
-  // Parse the exponent
-  //
-  if ((p != src_end) && (*p == 'e' || *p == 'E')) {
-    p++;
-    if(p == src_end) { return NUMBER_ERROR; }
-    bool exp_neg = *p == '-';
-    p += exp_neg || *p == '+';

-    uint64_t exp = 0;
-    const uint8_t *start_exp_digits = p;
-    while ((p != src_end) && parse_digit(*p, exp)) { p++; }
-    // no exp digits, or 20+ exp digits
-    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+    if (atomparsing::is_valid_nan_in_string(src)) {
+      return std::numeric_limits<float>::quiet_NaN();
+    }
+#endif

-    exponent += exp_neg ? 0-exp : exp;
+    return INCORRECT_TYPE;
   }
-
-  if ((p != src_end) && jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
-
-  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
-
-  //
-  // Assemble (or slow-parse) the float
-  //
-  double d;
-  if (simdjson_likely(!overflow)) {
-    if (compute_float_64(exponent, i, negative, d)) { return d; }
-  }
-  if (!parse_float_fallback(src - uint8_t(negative), src_end, &d)) {
-    return NUMBER_ERROR;
-  }
-  return d;
-}
-
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * src) noexcept {
-  //
-  // Check for minus sign
-  //
-  bool negative = (*(src + 1) == '-');
-  src += uint8_t(negative) + 1;
-
-  //
-  // Parse the integer part.
-  //
-  uint64_t i = 0;
-  const uint8_t *p = src;
-  p += parse_digit(*p, i);
-  bool leading_zero = (i == 0);
-  while (parse_digit(*p, i)) { p++; }
-  // no integer digits, or 0123 (zero must be solo)
-  if ( p == src ) { return INCORRECT_TYPE; }
   if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }

   //
@@ -50917,9 +63592,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   if (simdjson_likely(*p == '.')) {
     p++;
     const uint8_t *start_decimal_digits = p;
-    if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
-    p++;
-    while (parse_digit(*p, i)) { p++; }
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
     exponent = -(p - start_decimal_digits);

     // Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -50958,9 +63632,9 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   //
   // Assemble (or slow-parse) the float
   //
-  double d;
+  float d;
   if (simdjson_likely(!overflow)) {
-    if (compute_float_64(exponent, i, negative, d)) { return d; }
+    if (compute_float_32(exponent, i, negative, d)) { return d; }
   }
   if (!parse_float_fallback(src - uint8_t(negative), &d)) {
     return NUMBER_ERROR;
@@ -51251,10 +63925,6 @@ simdjson_inline int count_ones(uint64_t input_num) {
   return __lsx_vpickve2gr_w(__lsx_vpcnt_d(__m128i(v2u64{input_num, 0})), 0);
 }

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2, uint64_t *result) {
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-}

 } // unnamed namespace
 } // namespace lsx
@@ -52780,6 +65450,296 @@ simdjson_inline uint32_t find_next_document_index(dom_parser_implementation &par
   return 0;
 }

+/**
+ * Sentinel value returned to indicate a document started but didn't fit
+ * (CAPACITY error), as opposed to 0 which means no document content found
+ * (EMPTY).
+ */
+constexpr uint32_t DOCUMENT_TOO_LARGE = UINT32_MAX;
+
+/**
+ * For RFC 7464 JSON text sequences, filter RS from structural indexes and
+ * find batch boundaries.
+ *
+ * In JSON sequence mode, RS (0x1E) marks the start of each JSON text.
+ * RS bytes appear in structural_indexes as they are classified as scalars.
+ * This function:
+ * 1. Scans structural_indexes to find and count RS positions
+ * 2. Filters RS out of structural_indexes in-place
+ * 3. Determines batch boundaries based on RS positions
+ *
+ * @param parser The parser with structural_indexes and buf.
+ * @param len The length of the current batch buffer.
+ * @param is_final True if this is the final batch (no more data coming).
+ * @param next_batch_start Output: offset where the next batch should start.
+ * @param scan_len Offset past which no value start may be derived. Equals len
+ *        unless stage 1 dropped a trailing unclosed string, whose bytes it
+ *        never validated.
+ * @return The number of structural indexes to keep (after RS filtering),
+ *         0 if no document content found (EMPTY),
+ *         or DOCUMENT_TOO_LARGE if a document started but didn't fit (CAPACITY).
+ */
+simdjson_inline uint32_t find_next_document_index_json_sequence(
+    dom_parser_implementation &parser,
+    size_t len,
+    bool is_final,
+    uint32_t &next_batch_start,
+    size_t scan_len) {
+  // Default: next batch starts at end of buffer
+  next_batch_start = uint32_t(len);
+
+  if (parser.n_structural_indexes == 0) { return 0; }
+
+  // Phase 1: Scan structural_indexes to find RS positions and handle them.
+  // RS marks the start of a JSON text. For objects/arrays, the '{' or '[' after RS
+  // is already in structural_indexes (it's an operator). For scalars like numbers,
+  // the digit following RS is NOT in structural_indexes because the scanner sees
+  // RS as a scalar, making the digit a scalar continuation, not a start.
+  // We must: (1) remove RS from structural_indexes, and (2) for scalars, add the
+  // actual value start position.
+  // The EOF sentinel: len, or where a discarded unclosed string starts.
+  const uint32_t sentinel = parser.structural_indexes[parser.n_structural_indexes];
+  uint32_t write_idx = 0;
+  uint32_t last_rs_pos = 0;
+  uint32_t rs_count = 0;
+
+  for (uint32_t read_idx = 0; read_idx < parser.n_structural_indexes; read_idx++) {
+    const uint32_t pos = parser.structural_indexes[read_idx];
+    if (parser.buf[pos] == 0x1E) {
+      // This is an RS character - find the actual JSON value start.
+      last_rs_pos = pos;
+      rs_count++;
+      // Skip past this RS and any whitespace *and any additional RSes*
+      // to locate the real value. Consecutive RSes are degenerate
+      // "empty records" per RFC 7464; we collapse them here. They do
+      // not always appear as separate entries in structural_indexes
+      // because the scanner groups runs of adjacent non-whitespace
+      // scalar bytes (including RS) into a single scalar start.
+      uint32_t value_start = pos + 1;
+      while (value_start < scan_len) {
+        const uint8_t c = parser.buf[value_start];
+        if (c == ' ' || c == '\t' || c == '\n' || c == '\r') {
+          value_start++;
+        } else if (c == 0x1E) {
+          // Collapsed empty record. Still count it so rs_count reflects
+          // the true number of record markers and last_rs_pos tracks
+          // the final one.
+          last_rs_pos = value_start;
+          rs_count++;
+          value_start++;
+        } else {
+          break;
+        }
+      }
+      // If the scanner emitted additional structurals inside the
+      // whitespace+RS run we just walked over (i.e., isolated RSes
+      // separated by whitespace), skip past them so we do not
+      // double-count or double-emit.
+      while (read_idx + 1 < parser.n_structural_indexes &&
+             parser.structural_indexes[read_idx + 1] < value_start) {
+        read_idx++;
+      }
+      // Check if the value start is an operator (always present in
+      // scanner structural_indexes) or a scalar-like start (which may
+      // be missing from structural_indexes and must be added here).
+      // Note: '"' is NOT always in structural_indexes. The scanner
+      // classifies '"' as a scalar character and emits it as a
+      // structural only when it is a *scalar start* (preceded by
+      // whitespace or an operator). When '"' immediately follows an
+      // RS (which the scanner also classifies as scalar), it is
+      // treated as a scalar continuation and not emitted - so we
+      // must add it here just like any other scalar value.
+      if (value_start < scan_len) {
+        const uint8_t c = parser.buf[value_start];
+        const bool is_operator =
+            (c == '{' || c == '}' || c == '[' || c == ']' ||
+             c == ':' || c == ',');
+        // If the next scanner structural is exactly at value_start,
+        // the scanner already emitted it (it followed whitespace) and
+        // we must not add a duplicate - a subsequent iteration will
+        // copy it into write_idx.
+        const bool already_emitted =
+            (read_idx + 1 < parser.n_structural_indexes &&
+             parser.structural_indexes[read_idx + 1] == value_start);
+        if (!is_operator && !already_emitted) {
+          // Scalar value (number/true/false/null/string) - add its
+          // position since scanner missed it.
+          parser.structural_indexes[write_idx++] = value_start;
+        }
+      }
+    } else {
+      // Not RS, copy to output
+      parser.structural_indexes[write_idx++] = pos;
+    }
+  }
+
+  // Update structural index count. Compaction left a stale index in the slot
+  // past the end: restore the EOF sentinel that stage 1 had planted there, which
+  // document_stream::truncated_bytes() reads after a final batch.
+  parser.n_structural_indexes = write_idx;
+  parser.structural_indexes[write_idx] = sentinel;
+
+  if (parser.n_structural_indexes == 0) {
+    // Only RS markers here: the last one opens a record continuing past the
+    // window, so that is where the next batch resumes.
+    if (!is_final && rs_count > 0) { next_batch_start = last_rs_pos; }
+    return 0;
+  }
+  if (rs_count == 0) {
+    // No RS found; for final batch, try generic boundary detection
+    return is_final ? find_next_document_index(parser) : 0;
+  }
+
+  // Phase 2: Determine batch boundaries based on RS positions
+
+  if (is_final) {
+    // Final batch: all documents are complete (last one ends at EOF).
+    // In json_sequence mode, RS markers define document boundaries, so all
+    // remaining structurals form complete documents. Return them all directly.
+    // (Calling find_next_document_index() would fail for scalar documents.)
+    return parser.n_structural_indexes;
+  }
+
+  // Partial batch: need to find complete documents only.
+  // A document starting at an RS is complete if there is another RS after it.
+  next_batch_start = last_rs_pos;
+
+  if (rs_count < 2) {
+    // Only one RS, so we have at most one document that may be incomplete.
+    // We cannot confirm it is complete without another RS.
+    // Return DOCUMENT_TOO_LARGE if content was found (write_idx > 0), 0 if only separators.
+    return (parser.n_structural_indexes > 0) ? DOCUMENT_TOO_LARGE : 0;
+  }
+
+  // We have at least 2 RS markers. The last complete document ends before last_rs_pos.
+
+  // Find the structural index cutoff: keep only structurals < last_rs_pos.
+  // Since we already filtered RS, all remaining structurals are valid.
+  // We iterate backward to find the last structural before last_rs_pos.
+  uint32_t keep_count = 0;
+  for (uint32_t i = parser.n_structural_indexes; i > 0; i--) {
+    if (parser.structural_indexes[i - 1] < last_rs_pos) {
+      keep_count = i;
+      break;
+    }
+  }
+
+  // No structurals before the last RS - no complete documents
+  if (keep_count == 0) { return 0; }
+
+  // All documents before the last RS are complete by definition (the next RS
+  // confirms their end). No need to call find_next_document_index() which
+  // would fail for scalar documents like `1` or `"hello"`.
+  return keep_count;
+}
+
+/**
+ * Filter comma-delimited documents by removing root-level commas from
+ * structural indexes.
+ *
+ * For comma-delimited format like `{...},{...},{...}`, we need to remove
+ * the commas that separate documents (depth 0) while preserving commas
+ * inside arrays and objects (depth > 0).
+ *
+ * After filtering, the structural indexes look like whitespace-delimited
+ * documents, so find_next_document_index() works unchanged.
+ *
+ * @param parser The parser with structural_indexes and buf.
+ * @param len The length of the current batch buffer.
+ * @param is_final True if this is the final batch (no more data coming).
+ * @param next_batch_start Output: offset where the next batch should start.
+ * @return The number of structural indexes to keep,
+ *         0 if no document content found (EMPTY),
+ *         or DOCUMENT_TOO_LARGE if a document started but didn't fit (CAPACITY).
+ */
+simdjson_inline uint32_t filter_comma_delimited(
+    dom_parser_implementation &parser,
+    size_t len,
+    bool is_final,
+    uint32_t &next_batch_start) {
+  // Default: next batch starts at end of buffer
+  next_batch_start = uint32_t(len);
+
+  if (parser.n_structural_indexes == 0) { return 0; }
+
+  // Track depth to identify root-level commas (depth 0)
+  int depth = 0;
+  // The EOF sentinel: len, or where a discarded unclosed string starts.
+  const uint32_t sentinel = parser.structural_indexes[parser.n_structural_indexes];
+  uint32_t write_idx = 0;
+  uint32_t last_root_comma_pos = 0;
+  uint32_t root_comma_count = 0;
+
+  for (uint32_t i = 0; i < parser.n_structural_indexes; i++) {
+    uint32_t idx = parser.structural_indexes[i];
+    uint8_t c = parser.buf[idx];
+
+    switch (c) {
+      case '{': case '[':
+        depth++;
+        parser.structural_indexes[write_idx++] = idx;
+        break;
+      case '}': case ']':
+        depth--;
+        parser.structural_indexes[write_idx++] = idx;
+        break;
+      case ',':
+        if (depth == 0) {
+          // Root-level comma = document boundary, skip it
+          last_root_comma_pos = idx;
+          root_comma_count++;
+          continue;
+        }
+        parser.structural_indexes[write_idx++] = idx;
+        break;
+      default:
+        // Colons, scalars, etc.
+        parser.structural_indexes[write_idx++] = idx;
+        break;
+    }
+  }
+
+  // Update structural index count. Compaction left a stale index in the slot
+  // past the end: restore the EOF sentinel that stage 1 had planted there, which
+  // document_stream::truncated_bytes() reads after a final batch.
+  parser.n_structural_indexes = write_idx;
+  parser.structural_indexes[write_idx] = sentinel;
+
+  if (parser.n_structural_indexes == 0) { return 0; }
+
+  if (is_final) {
+    // Final batch: use standard boundary detection on filtered indexes
+    return find_next_document_index(parser);
+  }
+
+  // Partial batch: need to find complete documents only.
+  // A document ending with a root comma is complete.
+  if (root_comma_count == 0) {
+    // No root commas found; we cannot confirm any document is complete.
+    // The whole batch might be one incomplete document.
+    // Return DOCUMENT_TOO_LARGE if content was found (write_idx > 0), 0 if only commas.
+    return (parser.n_structural_indexes > 0) ? DOCUMENT_TOO_LARGE : 0;
+  }
+
+  // We have at least one root comma. Documents before the last comma are complete.
+  next_batch_start = last_root_comma_pos + 1;
+
+  // Find the structural index cutoff: keep only structurals < last_root_comma_pos
+  uint32_t keep_count = 0;
+  for (uint32_t i = parser.n_structural_indexes; i > 0; i--) {
+    if (parser.structural_indexes[i - 1] < last_root_comma_pos) {
+      keep_count = i;
+      break;
+    }
+  }
+
+  if (keep_count == 0) { return 0; }
+
+  // Use standard boundary detection on the complete portion
+  parser.n_structural_indexes = keep_count;
+  return find_next_document_index(parser);
+}
+
 } // namespace stage1
 } // unnamed namespace
 } // namespace lsx
@@ -53213,7 +66173,6 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
         return EMPTY;
       }
     }
-
     parser.n_structural_indexes = new_structural_indexes;
   } else if (partial == stage1_mode::streaming_final) {
     if(have_unclosed_string) { parser.n_structural_indexes--; }
@@ -53241,6 +66200,75 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
         // the trailing garbage.
         return EMPTY;
     }
+  } else if (partial == stage1_mode::json_sequence_partial) {
+    // RFC 7464: use RS positions for batch boundaries
+    // A discarded unclosed string also caps how far the filter may scan.
+    size_t scan_len = len;
+    if(have_unclosed_string) {
+      parser.n_structural_indexes--;
+      if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return CAPACITY; }
+      scan_len = parser.structural_indexes[parser.n_structural_indexes];
+    }
+    uint32_t next_batch_start = uint32_t(len);
+    auto new_structural_indexes = find_next_document_index_json_sequence(parser, len, false, next_batch_start, scan_len);
+    if (new_structural_indexes == DOCUMENT_TOO_LARGE) {
+      return CAPACITY;
+    }
+    if (new_structural_indexes == 0) {
+      // An EMPTY batch must still advance next_batch_start, or document_stream
+      // re-parses the same bytes forever. CAPACITY when it cannot.
+      if (next_batch_start == 0) { return CAPACITY; }
+      parser.n_structural_indexes = 0;
+      parser.structural_indexes[0] = next_batch_start;
+      return EMPTY;
+    }
+    parser.n_structural_indexes = new_structural_indexes;
+    parser.structural_indexes[parser.n_structural_indexes] = next_batch_start;
+  } else if (partial == stage1_mode::json_sequence_final) {
+    // RFC 7464: final batch, last document extends to EOF
+    // As above: a discarded unclosed string caps how far the filter may scan.
+    size_t scan_len = len;
+    if(have_unclosed_string) {
+      parser.n_structural_indexes--;
+      scan_len = parser.structural_indexes[parser.n_structural_indexes];
+    }
+    uint32_t next_batch_start = uint32_t(len);
+    parser.n_structural_indexes = find_next_document_index_json_sequence(parser, len, true, next_batch_start, scan_len);
+    // The filter compacted structural_indexes in place and restored the EOF
+    // sentinel past the compacted end, so the copy below is either the start
+    // of a truncated document or len, as in streaming_final.
+    parser.structural_indexes[parser.n_structural_indexes + 1] = parser.structural_indexes[parser.n_structural_indexes];
+    parser.structural_indexes[parser.n_structural_indexes] = uint32_t(len);
+    if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return EMPTY; }
+  } else if (partial == stage1_mode::comma_delimited_partial) {
+    // Comma-delimited: filter root-level commas, use comma positions for batch boundaries
+    if(have_unclosed_string) {
+      parser.n_structural_indexes--;
+      if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return CAPACITY; }
+    }
+    uint32_t next_batch_start = uint32_t(len);
+    auto new_structural_indexes = filter_comma_delimited(parser, len, false, next_batch_start);
+    if (new_structural_indexes == DOCUMENT_TOO_LARGE) {
+      return CAPACITY;
+    }
+    if (new_structural_indexes == 0) {
+      // An EMPTY batch must still advance next_batch_start, or document_stream
+      // re-parses the same bytes forever. CAPACITY when it cannot.
+      if (next_batch_start == 0) { return CAPACITY; }
+      parser.n_structural_indexes = 0;
+      parser.structural_indexes[0] = next_batch_start;
+      return EMPTY;
+    }
+    parser.n_structural_indexes = new_structural_indexes;
+    parser.structural_indexes[parser.n_structural_indexes] = next_batch_start;
+  } else if (partial == stage1_mode::comma_delimited_final) {
+    // Comma-delimited: final batch, last document extends to EOF
+    if(have_unclosed_string) { parser.n_structural_indexes--; }
+    uint32_t next_batch_start = uint32_t(len);
+    parser.n_structural_indexes = filter_comma_delimited(parser, len, true, next_batch_start);
+    parser.structural_indexes[parser.n_structural_indexes + 1] = parser.structural_indexes[parser.n_structural_indexes];
+    parser.structural_indexes[parser.n_structural_indexes] = uint32_t(len);
+    if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return EMPTY; }
   }
   checker.check_eof();
   return checker.errors();
@@ -53323,7 +66351,6 @@ namespace {
 namespace stage2 {

 class json_iterator;
-class structural_iterator;
 struct tape_builder;
 struct tape_writer;

@@ -53620,7 +66647,7 @@ public:
    *
    * - increment_count(iter) - each time a value is found in an array or object.
    */
-  template<bool STREAMING, typename V>
+  template<bool STREAMING, bool UNPADDED, typename V>
   simdjson_warn_unused simdjson_inline error_code walk_document(V &visitor) noexcept;

   /**
@@ -53637,6 +66664,7 @@ public:
    *
    * They may include invalid JSON as well (such as `1.2.3` or `ture`).
    */
+  template<bool UNPADDED>
   simdjson_inline const uint8_t *peek() const noexcept;
   /**
    * Advance to the next token.
@@ -53645,6 +66673,7 @@ public:
    *
    * They may include invalid JSON as well (such as `1.2.3` or `ture`).
    */
+  template<bool UNPADDED>
   simdjson_inline const uint8_t *advance() noexcept;
   /**
    * Get the remaining length of the document, from the start of the current token.
@@ -53693,7 +66722,7 @@ public:
   simdjson_warn_unused simdjson_inline error_code visit_primitive(V &visitor, const uint8_t *value) noexcept;
 };

-template<bool STREAMING, typename V>
+template<bool STREAMING, bool UNPADDED, typename V>
 simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &visitor) noexcept {
   logger::log_start();

@@ -53708,7 +66737,7 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &
   // Read first value
   //
   {
-    auto value = advance();
+    auto value = advance<UNPADDED>();

     // Make sure the outer object or array is closed before continuing; otherwise, there are ways we
     // could get into memory corruption. See https://github.com/simdjson/simdjson/issues/906
@@ -53720,8 +66749,8 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &
     }

     switch (*value) {
-      case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
-      case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+      case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+      case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
       default: SIMDJSON_TRY( visitor.visit_root_primitive(*this, value) ); break;
     }
   }
@@ -53738,29 +66767,29 @@ object_begin:
   SIMDJSON_TRY( visitor.visit_object_start(*this) );

   {
-    auto key = advance();
+    auto key = advance<UNPADDED>();
     if (*key != '"') { log_error("Object does not start with a key"); return TAPE_ERROR; }
     SIMDJSON_TRY( visitor.increment_count(*this) );
     SIMDJSON_TRY( visitor.visit_key(*this, key) );
   }

 object_field:
-  if (simdjson_unlikely( *advance() != ':' )) { log_error("Missing colon after key in object"); return TAPE_ERROR; }
+  if (simdjson_unlikely( *advance<UNPADDED>() != ':' )) { log_error("Missing colon after key in object"); return TAPE_ERROR; }
   {
-    auto value = advance();
+    auto value = advance<UNPADDED>();
     switch (*value) {
-      case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
-      case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+      case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+      case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
       default: SIMDJSON_TRY( visitor.visit_primitive(*this, value) ); break;
     }
   }

 object_continue:
-  switch (*advance()) {
+  switch (*advance<UNPADDED>()) {
     case ',':
       SIMDJSON_TRY( visitor.increment_count(*this) );
       {
-        auto key = advance();
+        auto key = advance<UNPADDED>();
         if (simdjson_unlikely( *key != '"' )) { log_error("Key string missing at beginning of field in object"); return TAPE_ERROR; }
         SIMDJSON_TRY( visitor.visit_key(*this, key) );
       }
@@ -53788,16 +66817,16 @@ array_begin:

 array_value:
   {
-    auto value = advance();
+    auto value = advance<UNPADDED>();
     switch (*value) {
-      case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
-      case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+      case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+      case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
       default: SIMDJSON_TRY( visitor.visit_primitive(*this, value) ); break;
     }
   }

 array_continue:
-  switch (*advance()) {
+  switch (*advance<UNPADDED>()) {
     case ',': SIMDJSON_TRY( visitor.increment_count(*this) ); goto array_value;
     case ']': log_end_value("array"); SIMDJSON_TRY( visitor.visit_array_end(*this) ); goto scope_end;
     default: log_error("Missing comma between array values"); return TAPE_ERROR;
@@ -53825,11 +66854,29 @@ simdjson_inline json_iterator::json_iterator(dom_parser_implementation &_dom_par
     dom_parser{_dom_parser} {
 }

+// Stage 1 leaves a sentinel index of value `len`; reading buf[len] requires
+// padding. For UNPADDED, return a dummy so walk_document can fail cleanly (#2815).
+template<bool UNPADDED>
 simdjson_inline const uint8_t *json_iterator::peek() const noexcept {
-  return &buf[*(next_structural)];
+  const uint32_t idx = *(next_structural);
+  SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+    if (simdjson_unlikely(idx >= dom_parser.len)) {
+      static constexpr uint8_t unpadded_eof_sentinel = 0;
+      return &unpadded_eof_sentinel;
+    }
+  }
+  return &buf[idx];
 }
+template<bool UNPADDED>
 simdjson_inline const uint8_t *json_iterator::advance() noexcept {
-  return &buf[*(next_structural++)];
+  const uint32_t idx = *(next_structural++);
+  SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+    if (simdjson_unlikely(idx >= dom_parser.len)) {
+      static constexpr uint8_t unpadded_eof_sentinel = 0;
+      return &unpadded_eof_sentinel;
+    }
+  }
+  return &buf[idx];
 }
 simdjson_inline size_t json_iterator::remaining_len() const noexcept {
   return dom_parser.len - *(next_structural-1);
@@ -53869,7 +66916,20 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::visit_root_primit
     case '"': return visitor.visit_root_string(*this, value);
     case 't': return visitor.visit_root_true_atom(*this, value);
     case 'f': return visitor.visit_root_false_atom(*this, value);
+#if SIMDJSON_ENABLE_NAN_INF
+    case 'n': {
+      auto err = visitor.visit_root_null_atom(*this, value);
+      if (err == SUCCESS) { return err; }
+      // propagate the error value returned by a bad 'null' atom if parsing 'nan' fails
+      return visitor.visit_root_nan_atom(*this, value, err);
+    }
+    // 'N' isn't a canonically recognized atom, so we return a TAPE_ERROR if failure occurs
+    case 'N': return visitor.visit_root_nan_atom(*this, value, TAPE_ERROR);
+    case 'i':
+    case 'I': return visitor.visit_root_inf_atom(*this, value);
+#else
     case 'n': return visitor.visit_root_null_atom(*this, value);
+#endif
     case '-':
     case '0': case '1': case '2': case '3': case '4':
     case '5': case '6': case '7': case '8': case '9':
@@ -53891,7 +66951,20 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::visit_primitive(V
   switch (*value) {
     case 't': return visitor.visit_true_atom(*this, value);
     case 'f': return visitor.visit_false_atom(*this, value);
+#if SIMDJSON_ENABLE_NAN_INF
+    case 'n': {
+      auto err = visitor.visit_null_atom(*this, value);
+      if (err == SUCCESS) { return err; }
+      // propagate the error value returned by a bad 'null' atom if parsing 'nan' fails
+      return visitor.visit_nan_atom(*this, value, err);
+    }
+    // 'N' isn't a canonically recognized atom, so we return a TAPE_ERROR if failure occurs
+    case 'N': return visitor.visit_nan_atom(*this, value, TAPE_ERROR);
+    case 'i':
+    case 'I': return visitor.visit_inf_atom(*this, value);
+#else
     case 'n': return visitor.visit_null_atom(*this, value);
+#endif
     default:
       log_error("Non-value found when value was expected!");
       return TAPE_ERROR;
@@ -54077,9 +67150,13 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_string(const uint8_t *src, u
            within the unicode codepoint handling code. */
         src += bs_dist;
         dst += bs_dist;
-        if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
-          return nullptr;
-        }
+        // Decode adjacent Unicode escapes without returning to the
+        // quote-and-backslash scanner between code points.
+        do {
+          if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
+            return nullptr;
+          }
+        } while (src[0] == '\\' && src[1] == 'u');
       } else {
         /* simple 1:1 conversion. Will eat bs_dist+2 characters in input and
          * write bs_dist+1 characters to output
@@ -54102,6 +67179,77 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_string(const uint8_t *src, u
   }
 }

+/**
+ * Bounds-safe variant of parse_string for input buffers that are NOT padded to
+ * len + SIMDJSON_PADDING bytes. `buf_end` is one past the last readable input
+ * byte (buf + len). It behaves exactly like parse_string while we are at least
+ * SIMDJSON_PADDING bytes away from buf_end (so every speculative SIMD read stays
+ * in bounds); once within the final SIMDJSON_PADDING bytes it copies the few
+ * remaining bytes into a space-padded scratch buffer and finishes with the
+ * regular parse_string. This keeps the delicate escape/Unicode handling in one
+ * place (the proven parse_string) rather than duplicating it.
+ *
+ * Correctness relies on stage 1 having validated the string, i.e. there is an
+ * unescaped closing quote within [src, buf_end); that quote is therefore inside
+ * the copied scratch, so parse_string finds it without running off the scratch.
+ */
+simdjson_warn_unused simdjson_inline uint8_t *parse_string_safe(const uint8_t *src, uint8_t *dst, bool allow_replacement, const uint8_t *buf_end) {
+  // Far from the end: identical to parse_string's loop. The guard uses
+  // SIMDJSON_PADDING (>= BYTES_PROCESSED) so copy_and_find never reads past
+  // buf_end; escape/Unicode look-aheads read within the string (before the
+  // closing quote, which is < buf_end), so they are in bounds here too.
+  // We add margin (+12) for handle_unicode_codepoint's worst-case lookahead:
+  // after seeing a high surrogate, it does hex_to_u32_nocheck on the immediate
+  // following bytes (+6 from the '\'), then (if it sees \u) another
+  // hex_to_u32_nocheck at +8..+11 relative to the backslash that started the
+  // escape. With bs_dist up to BYTES_PROCESSED-1 this reaches +11 from the
+  // chunk start. The +12 margin ensures that even on kernels where
+  // BYTES_PROCESSED == SIMDJSON_PADDING (e.g. icelake) the 4-byte read stays
+  // in-bounds. The scratch fallback (3*PAD) is already safe.
+  while (src + SIMDJSON_PADDING + 12 <= buf_end) {
+    auto b = backslash_and_quote{};
+    auto bs_quote = b.copy_and_find(src, dst);
+    if (bs_quote.has_quote_first()) {
+      return dst + bs_quote.quote_index();
+    }
+    if (bs_quote.has_backslash()) {
+      auto bs_dist = bs_quote.backslash_index();
+      uint8_t escape_char = src[bs_dist + 1];
+      if (escape_char == 'u') {
+        src += bs_dist;
+        dst += bs_dist;
+        if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
+          return nullptr;
+        }
+      } else {
+        uint8_t escape_result = escape_map[escape_char];
+        if (escape_result == 0u) {
+          return nullptr;
+        }
+        dst[bs_dist] = escape_result;
+        src += bs_dist + 2;
+        dst += bs_dist + 1;
+      }
+    } else {
+      src += backslash_and_quote::BYTES_PROCESSED;
+      dst += backslash_and_quote::BYTES_PROCESSED;
+    }
+  }
+  // Within the final SIMDJSON_PADDING bytes: copy what remains into a
+  // space-padded scratch (spaces are neither quote nor backslash, so they do not
+  // disturb matching) and let the regular parser finish from there. The closing
+  // quote is within `remaining` (< SIMDJSON_PADDING), so parse_string finds it in
+  // the chunk starting at some offset <= remaining and reads at most
+  // BYTES_PROCESSED (<= SIMDJSON_PADDING) further -- i.e. under 2*SIMDJSON_PADDING.
+  // We size at 3x for a comfortable margin (the unicode look-ahead reads a few
+  // extra bytes past an escape).
+  uint8_t scratch[SIMDJSON_PADDING * 3];
+  const size_t remaining = size_t(buf_end - src); // < SIMDJSON_PADDING
+  std::memset(scratch, ' ', sizeof(scratch));
+  std::memcpy(scratch, src, remaining);
+  return parse_string(scratch, dst, allow_replacement);
+}
+
 simdjson_warn_unused simdjson_inline uint8_t *parse_wobbly_string(const uint8_t *src, uint8_t *dst) {
   // It is not ideal that this function is nearly identical to parse_string.
   while (1) {
@@ -54156,73 +67304,6 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_wobbly_string(const uint8_t

 #endif // SIMDJSON_SRC_GENERIC_STAGE2_STRINGPARSING_H
 /* end file generic/stage2/stringparsing.h for lsx */
-/* including generic/stage2/structural_iterator.h for lsx: #include <generic/stage2/structural_iterator.h> */
-/* begin file generic/stage2/structural_iterator.h for lsx */
-#ifndef SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H
-
-/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
-/* amalgamation skipped (editor-only): #define SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H */
-/* amalgamation skipped (editor-only): #include <generic/stage2/base.h> */
-/* amalgamation skipped (editor-only): #include <simdjson/generic/dom_parser_implementation.h> */
-/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
-
-namespace simdjson {
-namespace lsx {
-namespace {
-namespace stage2 {
-
-class structural_iterator {
-public:
-  const uint8_t* const buf;
-  uint32_t *next_structural;
-  dom_parser_implementation &dom_parser;
-
-  // Start a structural
-  simdjson_inline structural_iterator(dom_parser_implementation &_dom_parser, size_t start_structural_index)
-    : buf{_dom_parser.buf},
-      next_structural{&_dom_parser.structural_indexes[start_structural_index]},
-      dom_parser{_dom_parser} {
-  }
-  // Get the buffer position of the current structural character
-  simdjson_inline const uint8_t* current() {
-    return &buf[*(next_structural-1)];
-  }
-  // Get the current structural character
-  simdjson_inline char current_char() {
-    return buf[*(next_structural-1)];
-  }
-  // Get the next structural character without advancing
-  simdjson_inline char peek_next_char() {
-    return buf[*next_structural];
-  }
-  simdjson_inline const uint8_t* peek() {
-    return &buf[*next_structural];
-  }
-  simdjson_inline const uint8_t* advance() {
-    return &buf[*(next_structural++)];
-  }
-  simdjson_inline char advance_char() {
-    return buf[*(next_structural++)];
-  }
-  simdjson_inline size_t remaining_len() {
-    return dom_parser.len - *(next_structural-1);
-  }
-
-  simdjson_inline bool at_end() {
-    return next_structural == &dom_parser.structural_indexes[dom_parser.n_structural_indexes];
-  }
-  simdjson_inline bool at_beginning() {
-    return next_structural == dom_parser.structural_indexes.get();
-  }
-};
-
-} // namespace stage2
-} // unnamed namespace
-} // namespace lsx
-} // namespace simdjson
-
-#endif // SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H
-/* end file generic/stage2/structural_iterator.h for lsx */
 /* including generic/stage2/tape_builder.h for lsx: #include <generic/stage2/tape_builder.h> */
 /* begin file generic/stage2/tape_builder.h for lsx */
 #ifndef SIMDJSON_SRC_GENERIC_STAGE2_TAPE_BUILDER_H
@@ -54245,12 +67326,8 @@ namespace lsx {
 namespace {
 namespace stage2 {

-struct tape_builder {
-  template<bool STREAMING>
-  simdjson_warn_unused static simdjson_inline error_code parse_document(
-    dom_parser_implementation &dom_parser,
-    dom::document &doc) noexcept;
-
+template <bool UNPADDED>
+struct tape_builder_impl {
   /** Called when a non-empty document starts. */
   simdjson_warn_unused simdjson_inline error_code visit_document_start(json_iterator &iter) noexcept;
   /** Called when a non-empty document ends without error. */
@@ -54303,88 +67380,130 @@ struct tape_builder {
   simdjson_warn_unused simdjson_inline error_code visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept;
   simdjson_warn_unused simdjson_inline error_code visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept;

+#if SIMDJSON_ENABLE_NAN_INF
+  simdjson_warn_unused simdjson_inline error_code visit_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept;
+  simdjson_warn_unused simdjson_inline error_code visit_root_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept;
+  // Attempts to parse 'inf' or 'infinity' (case insensitive). Because neither are canonical atoms,
+  // this returns a tape error on failure.
+  simdjson_warn_unused simdjson_inline error_code visit_inf_atom(json_iterator &iter, const uint8_t *value) noexcept;
+  simdjson_warn_unused simdjson_inline error_code visit_root_inf_atom(json_iterator &iter, const uint8_t *value) noexcept;
+#endif
+
   /** Called each time a new field or element in an array or object is found. */
   simdjson_warn_unused simdjson_inline error_code increment_count(json_iterator &iter) noexcept;

   /** Next location to write to tape */
   tape_writer tape;
+public:
+  simdjson_inline tape_builder_impl(dom::document &doc) noexcept;
 private:
   /** Next write location in the string buf for stage 2 parsing */
   uint8_t *current_string_buf_loc;

-  simdjson_inline tape_builder(dom::document &doc) noexcept;
-
   simdjson_inline uint32_t next_tape_index(json_iterator &iter) const noexcept;
   simdjson_inline void start_container(json_iterator &iter) noexcept;
   simdjson_warn_unused simdjson_inline error_code end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept;
   simdjson_warn_unused simdjson_inline error_code empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept;
   simdjson_inline uint8_t *on_start_string(json_iterator &iter) noexcept;
   simdjson_inline void on_end_string(uint8_t *dst) noexcept;
-}; // struct tape_builder
+}; // struct tape_builder_impl

-template<bool STREAMING>
-simdjson_warn_unused simdjson_inline error_code tape_builder::parse_document(
-    dom_parser_implementation &dom_parser,
-    dom::document &doc) noexcept {
-  dom_parser.doc = &doc;
-  json_iterator iter(dom_parser, STREAMING ? dom_parser.next_structural_index : 0);
-  tape_builder builder(doc);
-  return iter.walk_document<STREAMING>(builder);
-}
+// Thin, non-templated entry so each architecture's stage2() keeps calling
+// tape_builder::parse_document<STREAMING> unchanged. It chooses the bounds-safe
+// (unpadded) or the regular (padded) tape_builder_impl ONCE per document, so the
+// choice is a compile-time constant inside the walk: the padded path carries no
+// extra branch or load (see tape_builder_impl::visit_string).
+struct tape_builder {
+  template<bool STREAMING>
+  simdjson_warn_unused static simdjson_inline error_code parse_document(
+      dom_parser_implementation &dom_parser, dom::document &doc) noexcept {
+    dom_parser.doc = &doc;
+    json_iterator iter(dom_parser, STREAMING ? dom_parser.next_structural_index : 0);
+    if (dom_parser._unpadded) {
+      tape_builder_impl<true> builder(doc);
+      return iter.walk_document<STREAMING, true>(builder);
+    } else {
+      tape_builder_impl<false> builder(doc);
+      return iter.walk_document<STREAMING, false>(builder);
+    }
+  }
+};

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_primitive(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_primitive(json_iterator &iter, const uint8_t *value) noexcept {
   return iter.visit_root_primitive(*this, value);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_primitive(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_primitive(json_iterator &iter, const uint8_t *value) noexcept {
   return iter.visit_primitive(*this, value);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_empty_object(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_empty_object(json_iterator &iter) noexcept {
   return empty_container(iter, internal::tape_type::START_OBJECT, internal::tape_type::END_OBJECT);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_empty_array(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_empty_array(json_iterator &iter) noexcept {
   return empty_container(iter, internal::tape_type::START_ARRAY, internal::tape_type::END_ARRAY);
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_document_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_document_start(json_iterator &iter) noexcept {
   start_container(iter);
   return SUCCESS;
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_object_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_object_start(json_iterator &iter) noexcept {
   start_container(iter);
   return SUCCESS;
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_array_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_array_start(json_iterator &iter) noexcept {
   start_container(iter);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_object_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_object_end(json_iterator &iter) noexcept {
   return end_container(iter, internal::tape_type::START_OBJECT, internal::tape_type::END_OBJECT);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_array_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_array_end(json_iterator &iter) noexcept {
   return end_container(iter, internal::tape_type::START_ARRAY, internal::tape_type::END_ARRAY);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_document_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_document_end(json_iterator &iter) noexcept {
   constexpr uint32_t start_tape_index = 0;
   tape.append(start_tape_index, internal::tape_type::ROOT);
   tape_writer::write(iter.dom_parser.doc->tape[start_tape_index], next_tape_index(iter), internal::tape_type::ROOT);
   return SUCCESS;
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_key(json_iterator &iter, const uint8_t *key) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_key(json_iterator &iter, const uint8_t *key) noexcept {
   return visit_string(iter, key, true);
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::increment_count(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::increment_count(json_iterator &iter) noexcept {
   iter.dom_parser.open_containers[iter.depth].count++; // we have a key value pair in the object at parser.dom_parser.depth - 1
   return SUCCESS;
 }

-simdjson_inline tape_builder::tape_builder(dom::document &doc) noexcept : tape{doc.tape.get()}, current_string_buf_loc{doc.string_buf.get()} {}
+template <bool UNPADDED>
+simdjson_inline tape_builder_impl<UNPADDED>::tape_builder_impl(dom::document &doc) noexcept : tape{doc.tape.get()}, current_string_buf_loc{doc.string_buf.get()} {}

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_string(json_iterator &iter, const uint8_t *value, bool key) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_string(json_iterator &iter, const uint8_t *value, bool key) noexcept {
   iter.log_value(key ? "key" : "string");
   uint8_t *dst = on_start_string(iter);
-  dst = stringparsing::parse_string(value+1, dst, false); // We do not allow replacement when the escape characters are invalid.
+  // We do not allow replacement when the escape characters are invalid.
+  // UNPADDED is a compile-time constant chosen once per document by
+  // tape_builder::parse_document, so the padded build instantiates only the
+  // plain parse_string call below -- no runtime branch and no flag load.
+  SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+    dst = stringparsing::parse_string_safe(value+1, dst, false, iter.buf + iter.dom_parser.len);
+  } else {
+    dst = stringparsing::parse_string(value+1, dst, false);
+  }
   if (dst == nullptr) {
     iter.log_error("Invalid escape in string");
     return STRING_ERROR;
@@ -54393,27 +67512,48 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_string(json_
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_string(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_string(json_iterator &iter, const uint8_t *value) noexcept {
   return visit_string(iter, value);
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_number(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_number(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("number");
-  error_code err = numberparsing::parse_number(value, tape);
+  const uint8_t *num = value;
+  std::unique_ptr<uint8_t[]> copy{}; // keeps a padded copy of the tail alive when used
+  SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+    // numberparsing reads ahead in 8-byte blocks for floats
+    // (is_made_of_eight_digits_fast reads up to 7 bytes past the digits), so a
+    // number whose digits reach the final bytes of an unpadded buffer would read
+    // past it. *(next_structural) is the offset of the token following this
+    // number, hence an upper bound on where the digits end; when that is within
+    // SIMDJSON_PADDING of the end we parse from a space-padded copy of the tail
+    // (mirroring visit_root_number). This fires only for numbers near the end.
+    if (simdjson_unlikely(*(iter.next_structural) + SIMDJSON_PADDING > iter.dom_parser.len)) {
+      const size_t rl = iter.remaining_len(); // bytes from `value` to the end of the document
+      copy.reset(new (std::nothrow) uint8_t[rl + SIMDJSON_PADDING]);
+      if (copy.get() == nullptr) { return MEMALLOC; }
+      std::memcpy(copy.get(), value, rl);
+      std::memset(copy.get() + rl, ' ', SIMDJSON_PADDING);
+      num = copy.get();
+    }
+  }
+  error_code err = numberparsing::parse_number(num, tape);
   if (simdjson_unlikely(err == BIGINT_ERROR &&
       iter.dom_parser._number_as_string)) {
     // Write big integer to string buffer using the same format as strings.
     // Scan digits the same way parse_number does (skip optional '-', then digits).
-    const uint8_t *p = value;
+    const uint8_t *p = num;
     if (*p == '-') p++;
     while (numberparsing::is_digit(*p)) p++;
     // The digit run must be terminated by a structural or whitespace character; otherwise the
     // token is malformed (e.g. "123456789123456789123x").
     if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
-    size_t len = size_t(p - value);
+    size_t len = size_t(p - num);
     tape.append(current_string_buf_loc - iter.dom_parser.doc->string_buf.get(), internal::tape_type::BIGINT);
     uint8_t *dst = current_string_buf_loc + sizeof(uint32_t);
-    memcpy(dst, value, len);
+    memcpy(dst, num, len);
     dst += len;
     on_end_string(dst);
     return SUCCESS;
@@ -54421,7 +67561,8 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_number(json_
   return err;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_number(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_number(json_iterator &iter, const uint8_t *value) noexcept {
   //
   // We need to make a copy to make sure that the string is space terminated.
   // This is not about padding the input, which should already padded up
@@ -54435,76 +67576,142 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_number(
   // practice unless you are in the strange scenario where you have many JSON
   // documents made of single atoms.
   //
-  std::unique_ptr<uint8_t[]>copy(new (std::nothrow) uint8_t[iter.remaining_len() + SIMDJSON_PADDING]);
+  // In a stream, the input goes on with other documents: copy up to the next
+  // structural only, not to the end of the batch.
+  const size_t len = (std::min)(iter.remaining_len(), size_t(*iter.next_structural) - size_t(*(iter.next_structural - 1)));
+  std::unique_ptr<uint8_t[]>copy(new (std::nothrow) uint8_t[len + SIMDJSON_PADDING]);
   if (copy.get() == nullptr) { return MEMALLOC; }
-  std::memcpy(copy.get(), value, iter.remaining_len());
-  std::memset(copy.get() + iter.remaining_len(), ' ', SIMDJSON_PADDING);
+  std::memcpy(copy.get(), value, len);
+  std::memset(copy.get() + len, ' ', SIMDJSON_PADDING);
   error_code error = visit_number(iter, copy.get());
   return error;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("true");
-  if (!atomparsing::is_valid_true_atom(value)) { return T_ATOM_ERROR; }
+  // The non-length-aware validator reads a fixed 5 bytes; a malformed/truncated
+  // token at the very end of an unpadded buffer would over-read. Use the
+  // length-aware form there (the root variant already does this).
+  const bool ok = UNPADDED ? atomparsing::is_valid_true_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_true_atom(value);
+  if (!ok) { return T_ATOM_ERROR; }
   tape.append(0, internal::tape_type::TRUE_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("true");
   if (!atomparsing::is_valid_true_atom(value, iter.remaining_len())) { return T_ATOM_ERROR; }
   tape.append(0, internal::tape_type::TRUE_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("false");
-  if (!atomparsing::is_valid_false_atom(value)) { return F_ATOM_ERROR; }
+  const bool ok = UNPADDED ? atomparsing::is_valid_false_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_false_atom(value);
+  if (!ok) { return F_ATOM_ERROR; }
   tape.append(0, internal::tape_type::FALSE_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("false");
   if (!atomparsing::is_valid_false_atom(value, iter.remaining_len())) { return F_ATOM_ERROR; }
   tape.append(0, internal::tape_type::FALSE_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("null");
-  if (!atomparsing::is_valid_null_atom(value)) { return N_ATOM_ERROR; }
+  const bool ok = UNPADDED ? atomparsing::is_valid_null_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_null_atom(value);
+  if (!ok) { return N_ATOM_ERROR; }
   tape.append(0, internal::tape_type::NULL_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("null");
   if (!atomparsing::is_valid_null_atom(value, iter.remaining_len())) { return N_ATOM_ERROR; }
   tape.append(0, internal::tape_type::NULL_VALUE);
   return SUCCESS;
 }

+#if SIMDJSON_ENABLE_NAN_INF
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept {
+  iter.log_value("nan");
+  // For unpadded input use the length-aware validator so the 'infinity'-style
+  // 8-byte compare cannot read past the buffer on a malformed token at the end.
+  const bool ok = UNPADDED ? atomparsing::is_valid_nan_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_nan_atom(value);
+  if (!ok) { return errc; }
+  tape.append_double(std::numeric_limits<double>::quiet_NaN());
+  return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept {
+  iter.log_value("nan");
+  if (!atomparsing::is_valid_nan_atom(value, iter.remaining_len())) { return errc; }
+  tape.append_double(std::numeric_limits<double>::quiet_NaN());
+  return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_inf_atom(json_iterator &iter, const uint8_t *value) noexcept {
+  iter.log_value("inf");
+  // Because 'inf' is an extension, non a canonical atom, a tape error should be returned on failure.
+  // For unpadded input use the length-aware validator so the 'infinity' 8-byte
+  // compare cannot read past the buffer on a malformed token at the end.
+  const bool ok = UNPADDED ? atomparsing::is_valid_inf_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_inf_atom(value);
+  if (!ok) { return TAPE_ERROR; }
+  tape.append_double(std::numeric_limits<double>::infinity());
+  return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_inf_atom(json_iterator &iter, const uint8_t *value) noexcept {
+  iter.log_value("inf");
+  // Because 'inf' is an extension, non a canonical atom, a tape error should be returned on failure
+  if (!atomparsing::is_valid_inf_atom(value, iter.remaining_len())) { return TAPE_ERROR; }
+  tape.append_double(std::numeric_limits<double>::infinity());
+  return SUCCESS;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
 // private:

-simdjson_inline uint32_t tape_builder::next_tape_index(json_iterator &iter) const noexcept {
+template <bool UNPADDED>
+simdjson_inline uint32_t tape_builder_impl<UNPADDED>::next_tape_index(json_iterator &iter) const noexcept {
   return uint32_t(tape.next_tape_loc - iter.dom_parser.doc->tape.get());
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
   auto start_index = next_tape_index(iter);
   tape.append(start_index+2, start);
   tape.append(start_index, end);
   return SUCCESS;
 }

-simdjson_inline void tape_builder::start_container(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_inline void tape_builder_impl<UNPADDED>::start_container(json_iterator &iter) noexcept {
   iter.dom_parser.open_containers[iter.depth].tape_index = next_tape_index(iter);
   iter.dom_parser.open_containers[iter.depth].count = 0;
   tape.skip(); // We don't actually *write* the start element until the end.
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
   // Write the ending tape element, pointing at the start location
   const uint32_t start_tape_index = iter.dom_parser.open_containers[iter.depth].tape_index;
   tape.append(start_tape_index, end);
@@ -54517,13 +67724,15 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::end_container(json
   return SUCCESS;
 }

-simdjson_inline uint8_t *tape_builder::on_start_string(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_inline uint8_t *tape_builder_impl<UNPADDED>::on_start_string(json_iterator &iter) noexcept {
   // we advance the point, accounting for the fact that we have a NULL termination
   tape.append(current_string_buf_loc - iter.dom_parser.doc->string_buf.get(), internal::tape_type::STRING);
   return current_string_buf_loc + sizeof(uint32_t);
 }

-simdjson_inline void tape_builder::on_end_string(uint8_t *dst) noexcept {
+template <bool UNPADDED>
+simdjson_inline void tape_builder_impl<UNPADDED>::on_end_string(uint8_t *dst) noexcept {
   uint32_t str_length = uint32_t(dst - (current_string_buf_loc + sizeof(uint32_t)));
   // TODO check for overflow in case someone has a crazy string (>=4GB?)
   // But only add the overflow check when the document itself exceeds 4GB
@@ -54793,11 +68002,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
   return __builtin_popcountll(input_num);
 }

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
-                                uint64_t *result) {
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-}

 } // unnamed namespace
 } // namespace rvv_vls
@@ -55538,6 +68742,9 @@ namespace atomparsing {
 // to the compile-time constant 1936482662.
 simdjson_inline uint32_t string_to_uint32(const char* str) { uint32_t val; std::memcpy(&val, str, sizeof(uint32_t)); return val; }

+// Acts on the same principle as string_to_uint32, but on an 8-byte block of memory
+simdjson_inline uint64_t string_to_uint64(const char* str) { uint64_t val; std::memcpy(&val, str, sizeof(uint64_t)); return val; }
+

 // Again in str4ncmp we use a memcpy to avoid undefined behavior. The memcpy may appear expensive.
 // Yet all decent optimizing compilers will compile memcpy to a single instruction, just about.
@@ -55549,6 +68756,28 @@ simdjson_inline uint32_t str4ncmp(const uint8_t *src, const char* atom) {
   return srcval ^ string_to_uint32(atom);
 }

+// Checks that the first 8 characters of the input string match the given atom in a case-insensitive manner.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint64_t str8ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+  uint64_t srcval; // we want to avoid unaligned 32-bit loads (undefined in C/C++)
+  static_assert(sizeof(uint64_t) <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be larger than 8 bytes");
+  std::memcpy(&srcval, src, sizeof(uint64_t));
+
+  return (srcval | 0x2020202020202020ull) ^ string_to_uint64(atom);
+}
+
+// Checks that the first 3 characters of 'src' match 'atom' in a case-insensitive way.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint32_t str3ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+  return ((src[0] | 0x20) ^ atom[0]) //
+       | ((src[1] | 0x20) ^ atom[1]) //
+       | ((src[2] | 0x20) ^ atom[2]);
+}
+
 simdjson_warn_unused
 simdjson_inline bool is_valid_true_atom(const uint8_t *src) {
   return (str4ncmp(src, "true") | jsoncharutils::is_not_structural_or_whitespace(src[4])) == 0;
@@ -55585,6 +68814,71 @@ simdjson_inline bool is_valid_null_atom(const uint8_t *src, size_t len) {
   else { return false; }
 }

+#if SIMDJSON_ENABLE_NAN_INF
+// "nan" is 3 bytes; we check characters and then verify the next
+// character is structural or whitespace. We accept both "nan" and "NaN".
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src) {
+  return (str3ncmp_case_insensitive(src, "nan")
+        | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+}
+
+// checks that the next four characters of a string are 'nan"', where the 'nan'
+// is checked in a case-insensitive way.
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_in_string(const uint8_t *src) {
+  return (str3ncmp_case_insensitive(src, "nan") | (src[3] ^ '"')) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src, size_t len) {
+  if (len > 3) { return is_valid_nan_atom(src); }
+  if (len == 3) { return str3ncmp_case_insensitive(src, "nan") == 0; }
+  return false;
+}
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src) {
+  bool is_short_inf = (str3ncmp_case_insensitive(src, "inf")
+                     | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+  if(is_short_inf) return true;
+
+  // Check for 'infinity' (any capitalization)
+  return (str8ncmp_case_insensitive(src, "infinity") | jsoncharutils::is_not_structural_or_whitespace(src[8])) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_in_string(const uint8_t *src) {
+  bool is_short_inf = (str3ncmp_case_insensitive(src, "inf") | (src[3] ^ '"')) == 0;
+  if(is_short_inf) return true;
+
+  return (str8ncmp_case_insensitive(src, "infinity") | (src[8] ^ '"')) == 0;
+}
+
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src, size_t len) {
+  if (len > 8) { return is_valid_inf_atom(src); }
+  if (len == 8 && str8ncmp_case_insensitive(src, "infinity") == 0) {
+    return true;
+  }
+  if (len > 3) {
+    return (str3ncmp_case_insensitive(src, "inf")
+          | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+  }
+  if (len == 3) { return str3ncmp_case_insensitive(src, "inf") == 0; }
+  return false;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
 } // namespace atomparsing
 } // unnamed namespace
 } // namespace rvv_vls
@@ -55675,7 +68969,7 @@ inline simdjson_warn_unused error_code dom_parser_implementation::set_capacity(s
 }

 inline simdjson_warn_unused error_code dom_parser_implementation::set_max_depth(size_t max_depth) noexcept {
-  if(max_depth == 0 || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
+  if(max_depth == 0 || max_depth > SIMDJSON_MAX_DEPTH || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
   // Stage 2 stacks
   open_containers.reset(new (std::nothrow) open_container[max_depth]);
   is_array.reset(new (std::nothrow) bool[max_depth]);
@@ -55854,6 +69148,7 @@ protected:
 /* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_NUMBERPARSING_H */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/base.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/jsoncharutils.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/atomparsing.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/internal/numberparsing_tables.h" */
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */

@@ -55893,6 +69188,18 @@ simdjson_inline double to_double(uint64_t mantissa, uint64_t real_exponent, bool
     return d;
 }

+// Convert a mantissa, an exponent and a sign bit into an ieee32 float (binary32).
+// The real_exponent needs to be in [0, 254] (technically real_exponent = 255 would be acceptable).
+// The mantissa should be in [0,1<<24). The bit at index (1U << 23) will be zeroed.
+simdjson_inline float to_float(uint32_t mantissa, uint32_t real_exponent, bool negative) {
+    float f;
+    mantissa &= ~(uint32_t(1) << 23);
+    mantissa |= real_exponent << 23;
+    mantissa |= ((static_cast<uint32_t>(negative)) << 31);
+    std::memcpy(&f, &mantissa, sizeof(f));
+    return f;
+}
+
 // Attempts to compute i * 10^(power) exactly; and if "negative" is
 // true, negate the result.
 // This function will only work in some cases, when it does not work, success is
@@ -56149,6 +69456,219 @@ simdjson_inline bool compute_float_64(int64_t power, uint64_t i, bool negative,
   return true;
 }

+// Attempts to compute i * 10^(power) as a binary32 (float) value; and if
+// "negative" is true, negate the result.
+//
+// This is the same Eisel-Lemire algorithm as compute_float_64, retargeted at
+// binary32. It is adapted from fast_float
+// (https://github.com/fastfloat/fast_float), where the routine is written once
+// and instantiated for each binary format. We only need 24 bits of mantissa
+// instead of 53, so we ask the truncated product for 24 + 2 = 26 accurate bits
+// (plus one bit that may be lost to the "upperbit" shift), and we shift the
+// product right by (upperbit + 64 - 23 - 3) instead of (upperbit + 64 - 52 - 3).
+//
+// The power-of-five table is shared with the binary64 code: the useful range
+// for binary32, [smallest_power_binary32, largest_power_binary32], is a strict
+// subset of [smallest_power, largest_power].
+//
+// The function returns false when the result would be infinite: simdjson
+// refuses to parse infinite values, so the caller reports an error. Unlike
+// compute_float_64, a false return never means "try harder": the accuracy of
+// the two-limb product is guaranteed by Noble Mushtak and Daniel Lemire,
+// "Fast Number Parsing Without Fallback" (https://arxiv.org/abs/2212.06644).
+// The caller must still fall back to a slow path when the 64-bit mantissa i
+// was truncated (more than 19 significant digits).
+//
+// We assume that power is in the [smallest_power, largest_power] interval: the
+// caller is responsible for this check.
+simdjson_inline bool compute_float_32(int64_t power, uint64_t i, bool negative, float &d) {
+  // Powers of ten that are exactly representable as binary32 values: 10^k is
+  // exact as long as k <= 10 (5^10 = 9765625 < 2**24 while 5^11 > 2**24).
+  static constexpr float power_of_ten_float[] = {
+      1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+      1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+  // The range of powers of ten that a non-zero, finite binary32 value can be
+  // built from. Anything smaller rounds to zero, anything larger is infinite.
+  // These are the smallest_power_of_ten()/largest_power_of_ten() constants that
+  // fast_float uses for binary32.
+  constexpr int smallest_power_binary32 = -65;
+  constexpr int largest_power_binary32 = 38;
+
+  // We start with the fast path described in
+  // Clinger WD. How to read floating point numbers accurately.
+  // ACM SIGPLAN Notices. 1990
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+  // We cannot be certain that x/y is rounded to nearest.
+  if (0 <= power && power <= 10 && i <= 16777215)
+#else
+  if (-10 <= power && power <= 10 && i <= 16777215)
+#endif
+  {
+    // Convert the integer into a float. This is lossless since
+    // 0 <= i <= 2^24 - 1.
+    d = float(i);
+    // Both d and the power of ten are exactly representable as binary32
+    // values, so the product (or quotient) is correctly rounded.
+    if (power < 0) {
+      d = d / power_of_ten_float[-power];
+    } else {
+      d = d * power_of_ten_float[power];
+    }
+    if (negative) {
+      d = -d;
+    }
+    return true;
+  }
+
+  // The fast path has failed, so we fall back on the Eisel-Lemire algorithm.
+  // It needs i > 0 (so that the leading bit of i can be normalized), so we
+  // handle i == 0 separately. We also handle the powers of ten that are so
+  // small (or so large) that the answer is zero (or infinite) whatever the
+  // mantissa is.
+  if (i == 0 || power < smallest_power_binary32) {
+    d = negative ? -0.0f : 0.0f;
+    return true;
+  }
+  if (power > largest_power_binary32) {
+    // We have, for sure, an infinite value.
+    return false;
+  }
+
+  // The exponent is 128 + 63 + power + floor(log(5**power)/log(2)).
+  // The 128 comes from the ieee32 standard (the minimal exponent is -127).
+  // The 63 comes from the fact that we use a 64-bit word.
+  // See compute_float_64 for a discussion of the magical 152170 + 65536.
+  int64_t exponent = (((152170 + 65536) * power) >> 16) + 128 + 63;
+
+  // We want the most significant bit of i to be 1. Shift if needed.
+  int lz = leading_zeroes(i);
+  i <<= lz;
+
+  // We want the most significant 64 bits of the product i * 5**power. It is
+  // safe to index the table because
+  // smallest_power <= smallest_power_binary32 <= power
+  //               <= largest_power_binary32 <= largest_power.
+  const uint32_t index = 2 * uint32_t(power - simdjson::internal::smallest_power);
+#if SIMDJSON_STATIC_REFLECTION
+  simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index]);
+#else
+  simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index]);
+#endif
+
+  // Unless the least significant 38 bits of the high (64-bit) part of the full
+  // product are all 1s, then we know that the most significant 26 bits are
+  // exact and no further work is needed. Having 26 bits is necessary because
+  // we need 24 bits for the mantissa but we have to have one rounding bit and
+  // we can waste a bit if the most significant bit of the product is zero.
+  // (For binary64, the same reasoning gives 55 bits and a 9-bit mask.)
+  if((firstproduct.high & 0x3FFFFFFFFF) == 0x3FFFFFFFFF) {
+    // The truncated multiplication was not accurate enough; use the next 64
+    // bits of the power of five to refine it. See compute_float_64 for a
+    // detailed discussion.
+#if SIMDJSON_STATIC_REFLECTION
+    simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index + 1]);
+#else
+    simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index + 1]);
+#endif
+    firstproduct.low += secondproduct.high;
+    if(secondproduct.high > firstproduct.low) { firstproduct.high++; }
+  }
+  uint64_t lower = firstproduct.low;
+  uint64_t upper = firstproduct.high;
+  // The final mantissa should be 24 bits with a leading 1.
+  // We shift it so that it occupies 25 bits with a leading 1.
+  ///////
+  uint64_t upperbit = upper >> 63;
+  uint64_t mantissa = upper >> (upperbit + 38); // 38 == 64 - 23 - 3
+  lz += int(1 ^ upperbit);
+
+  // Here we have mantissa < (1<<25).
+  int64_t real_exponent = exponent - lz;
+  if (simdjson_unlikely(real_exponent <= 0)) { // we have a subnormal?
+    // Here we have that real_exponent <= 0 so -real_exponent >= 0
+    if(-real_exponent + 1 >= 64) { // if we have more than 64 bits below the minimum exponent, you have a zero for sure.
+      d = negative ? -0.0f : 0.0f;
+      return true;
+    }
+    // next line is safe because -real_exponent + 1 < 64
+    mantissa >>= -real_exponent + 1;
+    // Thankfully, we can't have both "round-to-even" and subnormals because
+    // "round-to-even" only occurs for powers close to 0.
+    mantissa += (mantissa & 1); // round up
+    mantissa >>= 1;
+    // As in compute_float_64, rounding up may take us out of the subnormal
+    // range, so we can only decide after rounding.
+    real_exponent = (mantissa < (uint64_t(1) << 23)) ? 0 : 1;
+    d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+    return true;
+  }
+  // We have to round to even. The "to even" part is only a problem when we are
+  // right in between two floats, which we guard against. The bounds on the
+  // power of ten are those of fast_float's
+  // min_exponent_round_to_even()/max_exponent_round_to_even() for binary32:
+  // when q >= 0, (2m+1) must be divisible by 5^q with 5^q <= 2^25, so q <= 10;
+  // when q < 0, we need 2^24 x 5^{-q} < 2^{64}, so q >= -17.
+  if (simdjson_unlikely((lower <= 1) && (power >= -17) && (power <= 10) && ((mantissa & 3) == 1))) {
+    if((mantissa << (upperbit + 38)) == upper) {
+      mantissa &= ~uint64_t(1);   // flip it so that we do not round up
+    }
+  }
+
+  mantissa += mantissa & 1;
+  mantissa >>= 1;
+
+  // Here we have mantissa < (1<<24), unless there was an overflow
+  if (mantissa >= (uint64_t(1) << 24)) {
+    mantissa = (uint64_t(1) << 23);
+    real_exponent++;
+  }
+  mantissa &= ~(uint64_t(1) << 23);
+  // we have to check that real_exponent is in range, otherwise we bail out
+  if (simdjson_unlikely(real_exponent > 254)) {
+    // We have an infinite value!!! We could actually throw an error here if we could.
+    return false;
+  }
+  d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+  return true;
+}
+
+#if SIMDJSON_ENABLE_NAN_INF
+// Parses a nan or infinity. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, double& d) noexcept {
+  if (atomparsing::is_valid_inf_atom(src)) {
+    double inf = std::numeric_limits<double>::infinity();
+    d = negative ? -inf : inf;
+    return true;
+  }
+
+  if (atomparsing::is_valid_nan_atom(src)) {
+    d = std::numeric_limits<double>::quiet_NaN();
+    return true;
+  }
+
+  return false;
+}
+
+// Parses a nan or infinity as a binary32 value. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, float& d) noexcept {
+  if (atomparsing::is_valid_inf_atom(src)) {
+    float inf = std::numeric_limits<float>::infinity();
+    d = negative ? -inf : inf;
+    return true;
+  }
+
+  if (atomparsing::is_valid_nan_atom(src)) {
+    d = std::numeric_limits<float>::quiet_NaN();
+    return true;
+  }
+
+  return false;
+}
+#endif
+
 // We call a fallback floating-point parser that might be slow. Note
 // it will accept JSON numbers, but the JSON spec. is more restrictive so
 // before you call parse_float_fallback, you need to have validated the input
@@ -56186,6 +69706,16 @@ static bool parse_float_fallback(const uint8_t *ptr, const uint8_t *end_ptr, dou
   return !(*outDouble > (std::numeric_limits<double>::max)() || *outDouble < std::numeric_limits<double>::lowest());
 }

+// Same as parse_float_fallback, but for binary32 (float) values. Going through
+// the binary64 fallback and then rounding to binary32 would be subject to
+// double rounding, so we run the fallback algorithm directly on binary32.
+static bool parse_float_fallback(const uint8_t *ptr, float *outFloat) {
+  *outFloat = simdjson::internal::from_chars_float(reinterpret_cast<const char *>(ptr));
+  // We do not accept infinite values. See the binary64 version above for why we
+  // do not use std::isfinite.
+  return !(*outFloat > (std::numeric_limits<float>::max)() || *outFloat < std::numeric_limits<float>::lowest());
+}
+
 // check quickly whether the next 8 chars are made of digits
 // at a glance, it looks better than Mula's
 // http://0x80.pl/articles/swar-digits-validate.html
@@ -56204,6 +69734,28 @@ simdjson_inline bool is_made_of_eight_digits_fast(const uint8_t *chars) {
           0x3333333333333333);
 }

+// The same idea for four characters. A block of eight only bites when eight
+// digits are left, so a run of 4 to 7 digits used to fall back to one digit at a
+// time; taking four of them at once is what makes the tail of a long fraction
+// cheap. Adapted from fast_float, credit @aqrit.
+simdjson_inline bool is_made_of_four_digits_fast(const uint8_t *chars) {
+  uint32_t val;
+  // Reads up to 3 bytes beyond the digits, which SIMDJSON_PADDING covers.
+  static_assert(3 <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be bigger than 3");
+  std::memcpy(&val, chars, 4);
+  return !(((val + 0x46464646) | (val - 0x30303030)) & 0x80808080);
+}
+
+// Only call this when is_made_of_four_digits_fast() says the four characters are
+// digits. Little-endian: guarded by SIMDJSON_SWAR_NUMBER_PARSING.
+simdjson_inline uint32_t parse_four_digits_unrolled(const uint8_t *chars) {
+  uint32_t val;
+  std::memcpy(&val, chars, 4);
+  val -= 0x30303030;
+  val = (val * 10) + (val >> 8);
+  return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
 template<typename I>
 SIMDJSON_NO_SANITIZE_UNDEFINED // We deliberately allow overflow here and check later
 simdjson_inline bool parse_digit(const uint8_t c, I &i) {
@@ -56220,26 +69772,73 @@ simdjson_inline bool is_digit(const uint8_t c) {
   return static_cast<uint8_t>(c - '0') <= 9;
 }

-simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
-  // we continue with the fiction that we have an integer. If the
-  // floating point number is representable as x * 10^z for some integer
-  // z that fits in 53 bits, then we will be able to convert back the
-  // the integer into a float in a lossless manner.
-  const uint8_t *const first_after_period = p;
+// Consumes a run of digits into i, eight at a time while eight are available,
+// then four, then one at a time. Overflow is deliberate: the caller counts the
+// digits and falls back when there are too many for a 64-bit mantissa.
+//
+// Long runs of digits are what a float-heavy document is made of: the fraction
+// of a binary64 printed to full precision is 15 to 17 digits, and reading those
+// one at a time is the single largest cost in parse_double(). Integer parts are
+// left alone: they are usually a handful of digits, and there the check for a
+// block of eight is wasted work.
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_fraction_digits(const uint8_t *&p, uint64_t &i) {
+#ifdef SIMDJSON_SWAR_NUMBER_PARSING
+#if SIMDJSON_SWAR_NUMBER_PARSING
+  while (is_made_of_eight_digits_fast(p)) {
+    i = i * 100000000 + parse_eight_digits_unrolled(p);
+    p += 8;
+  }
+  // A 4 to 7 digit remainder is the common case once the blocks of eight are
+  // gone: 15 fraction digits leave 7, and 17 leave 1 after two blocks.
+  if (is_made_of_four_digits_fast(p)) {
+    i = i * 10000 + parse_four_digits_unrolled(p);
+    p += 4;
+  }
+#endif // SIMDJSON_SWAR_NUMBER_PARSING
+#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
+  while (parse_digit(*p, i)) { p++; }
+}
+
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_float_integer_digits(const uint8_t *&p, uint64_t &i,
+                                                bool &leading_zero) {
+  if (!parse_digit(*p, i)) { leading_zero = true; return; }
+  p++;
+  leading_zero = (i == 0);
+  while (parse_digit(*p, i)) { p++; }
+}

+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_integer_digits(const uint8_t *&p, uint64_t &i) {
 #ifdef SIMDJSON_SWAR_NUMBER_PARSING
 #if SIMDJSON_SWAR_NUMBER_PARSING
-  // this helps if we have lots of decimals!
-  // this turns out to be frequent enough.
+  // Identifiers, timestamps and counters often have eight digits or more.
+  // Prior related work: jsonifier parses integers as eight-digit SWAR words
+  // (str_to_i.hpp, https://github.com/nihilai-collective/Jsonifier). This
+  // takes one such word with parse_eight_digits_unrolled, the routine
+  // simdjson uses for long fractions.
   if (is_made_of_eight_digits_fast(p)) {
     i = i * 100000000 + parse_eight_digits_unrolled(p);
     p += 8;
   }
+  const uint8_t *const swar_end = p + 8;
+  while (p < swar_end && is_made_of_four_digits_fast(p)) {
+    i = i * 10000 + parse_four_digits_unrolled(p);
+    p += 4;
+  }
 #endif // SIMDJSON_SWAR_NUMBER_PARSING
 #endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
-  // Unrolling the first digit makes a small difference on some implementations (e.g. westmere)
-  if (parse_digit(*p, i)) { ++p; }
   while (parse_digit(*p, i)) { p++; }
+}
+
+simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
+  // we continue with the fiction that we have an integer. If the
+  // floating point number is representable as x * 10^z for some integer
+  // z that fits in 53 bits, then we will be able to convert back the
+  // the integer into a float in a lossless manner.
+  const uint8_t *const first_after_period = p;
+  parse_fraction_digits(p, i);
   exponent = first_after_period - p;
   // Decimal without digits (123.) is illegal
   if (exponent == 0) {
@@ -56328,7 +69927,7 @@ simdjson_inline size_t significant_digits(const uint8_t * start_digits, size_t d
 } // unnamed namespace

 /** @private */
-static error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
+inline error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
   if (parse_float_fallback(src, answer)) {
     return SUCCESS;
   }
@@ -56411,15 +70010,17 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
   return SUCCESS;              // always succeeds
 }

-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept  { return false; }
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept  { return false; }
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept { return number_type::signed_integer; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline bool is_negative(const uint8_t *) noexcept  { return false; }
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t *) noexcept  { return false; }
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t *) noexcept { return number_type::signed_integer; }
 #else

 // parse the number at src
@@ -56450,7 +70051,21 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
   size_t digit_count = size_t(p - start_digits);
-  if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) { return INVALID_NUMBER(src); }
+  if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // By this point, we know that our input does not begin with a digit. We will attempt
+    // to handle NaN/Infinity.
+
+    double d;
+    if (compute_nan_inf(p, negative, d)) {
+      writer.append_double(d);
+      return SUCCESS;
+    }
+#endif
+
+    return INVALID_NUMBER(src);
+  }

   //
   // Handle floats if there is a . or e (or both)
@@ -56499,7 +70114,7 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
     // - Therefore, if the number is positive and lower than that, it's overflow.
     // - The value we are looking at is less than or equal to INT64_MAX.
     //
-    }  else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return INVALID_NUMBER(src); }
+    }  else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return BIGINT_NUMBER(src); }
   }

   // Write unsigned if it does not fit in a signed integer.
@@ -56598,7 +70213,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const u
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -56696,7 +70311,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_stri
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -56751,7 +70366,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uin
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -56837,7 +70452,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = src;
   uint64_t i = 0;
-  while (parse_digit(*src, i)) { src++; }
+  parse_integer_digits(src, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -56877,11 +70492,20 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   //
   uint64_t i = 0;
   const uint8_t *p = src;
-  p += parse_digit(*p, i);
-  bool leading_zero = (i == 0);
-  while (parse_digit(*p, i)) { p++; }
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
   // no integer digits, or 0123 (zero must be solo)
-  if ( p == src ) { return INCORRECT_TYPE; }
+  if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no loading digits, the number may be nan or infinity.
+    // Attempt to compute those, and return on success.
+    double d;
+    if (compute_nan_inf(p, negative, d)) { return d; }
+#endif
+
+    return INCORRECT_TYPE;
+  }
   if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }

   //
@@ -56892,9 +70516,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   if (simdjson_likely(*p == '.')) {
     p++;
     const uint8_t *start_decimal_digits = p;
-    if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
-    p++;
-    while (parse_digit(*p, i)) { p++; }
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
     exponent = -(p - start_decimal_digits);

     // Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -56943,6 +70566,97 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   return d;
 }

+// Parse a JSON number into a binary32 (float) value.
+//
+// This mirrors parse_double, but it rounds to binary32 directly instead of
+// rounding to binary64 and then to binary32: the latter is subject to double
+// rounding and would not always produce the nearest float.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * src) noexcept {
+  //
+  // Check for minus sign
+  //
+  bool negative = (*src == '-');
+  src += uint8_t(negative);
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, the number may be nan or infinity.
+    // Attempt to compute those, and return on success.
+    float f;
+    if (compute_nan_inf(p, negative, f)) { return f; }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = p-start_digits > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if (*p == 'e' || *p == 'E') {
+    p++;
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while (parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  float d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_32(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
 simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
   return (*src == '-');
 }
@@ -57095,11 +70809,25 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   //
   uint64_t i = 0;
   const uint8_t *p = src;
-  p += parse_digit(*p, i);
-  bool leading_zero = (i == 0);
-  while (parse_digit(*p, i)) { p++; }
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
   // no integer digits, or 0123 (zero must be solo)
-  if ( p == src ) { return INCORRECT_TYPE; }
+  if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, attempt to parse numbers that are either
+    // NaN or Infinity
+    if (atomparsing::is_valid_inf_in_string(src)) {
+      double inf = std::numeric_limits<double>::infinity();
+      return negative ? -inf : inf;
+    }
+
+    if (atomparsing::is_valid_nan_in_string(src)) {
+      return std::numeric_limits<double>::quiet_NaN();
+    }
+#endif
+
+    return INCORRECT_TYPE;
+  }
   if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }

   //
@@ -57110,9 +70838,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   if (simdjson_likely(*p == '.')) {
     p++;
     const uint8_t *start_decimal_digits = p;
-    if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
-    p++;
-    while (parse_digit(*p, i)) { p++; }
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
     exponent = -(p - start_decimal_digits);

     // Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -57161,6 +70888,99 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   return d;
 }

+// Parse a JSON number held inside a JSON string into a binary32 (float) value.
+// See parse_float for why we do not simply round parse_double_in_string.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * src) noexcept {
+  //
+  // Check for minus sign
+  //
+  bool negative = (*(src + 1) == '-');
+  src += uint8_t(negative) + 1;
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, attempt to parse numbers that are either
+    // NaN or Infinity
+    if (atomparsing::is_valid_inf_in_string(src)) {
+      float inf = std::numeric_limits<float>::infinity();
+      return negative ? -inf : inf;
+    }
+
+    if (atomparsing::is_valid_nan_in_string(src)) {
+      return std::numeric_limits<float>::quiet_NaN();
+    }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = p-start_digits > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if (*p == 'e' || *p == 'E') {
+    p++;
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while (parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if (*p != '"') { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  float d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_32(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
 } // unnamed namespace
 #endif // SIMDJSON_SKIPNUMBERPARSING

@@ -57339,7 +71159,7 @@ public:
   simdjson_inline implementation() : simdjson::implementation(
       "rvv_vls",
       "RISC-V V extension",
-      0
+      internal::instruction_set::RVV_VLS
   ) {}
   simdjson_warn_unused error_code create_dom_parser_implementation(
     size_t capacity,
@@ -57456,11 +71276,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
   return __builtin_popcountll(input_num);
 }

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
-                                uint64_t *result) {
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-}

 } // unnamed namespace
 } // namespace rvv_vls
@@ -59371,6 +73186,296 @@ simdjson_inline uint32_t find_next_document_index(dom_parser_implementation &par
   return 0;
 }

+/**
+ * Sentinel value returned to indicate a document started but didn't fit
+ * (CAPACITY error), as opposed to 0 which means no document content found
+ * (EMPTY).
+ */
+constexpr uint32_t DOCUMENT_TOO_LARGE = UINT32_MAX;
+
+/**
+ * For RFC 7464 JSON text sequences, filter RS from structural indexes and
+ * find batch boundaries.
+ *
+ * In JSON sequence mode, RS (0x1E) marks the start of each JSON text.
+ * RS bytes appear in structural_indexes as they are classified as scalars.
+ * This function:
+ * 1. Scans structural_indexes to find and count RS positions
+ * 2. Filters RS out of structural_indexes in-place
+ * 3. Determines batch boundaries based on RS positions
+ *
+ * @param parser The parser with structural_indexes and buf.
+ * @param len The length of the current batch buffer.
+ * @param is_final True if this is the final batch (no more data coming).
+ * @param next_batch_start Output: offset where the next batch should start.
+ * @param scan_len Offset past which no value start may be derived. Equals len
+ *        unless stage 1 dropped a trailing unclosed string, whose bytes it
+ *        never validated.
+ * @return The number of structural indexes to keep (after RS filtering),
+ *         0 if no document content found (EMPTY),
+ *         or DOCUMENT_TOO_LARGE if a document started but didn't fit (CAPACITY).
+ */
+simdjson_inline uint32_t find_next_document_index_json_sequence(
+    dom_parser_implementation &parser,
+    size_t len,
+    bool is_final,
+    uint32_t &next_batch_start,
+    size_t scan_len) {
+  // Default: next batch starts at end of buffer
+  next_batch_start = uint32_t(len);
+
+  if (parser.n_structural_indexes == 0) { return 0; }
+
+  // Phase 1: Scan structural_indexes to find RS positions and handle them.
+  // RS marks the start of a JSON text. For objects/arrays, the '{' or '[' after RS
+  // is already in structural_indexes (it's an operator). For scalars like numbers,
+  // the digit following RS is NOT in structural_indexes because the scanner sees
+  // RS as a scalar, making the digit a scalar continuation, not a start.
+  // We must: (1) remove RS from structural_indexes, and (2) for scalars, add the
+  // actual value start position.
+  // The EOF sentinel: len, or where a discarded unclosed string starts.
+  const uint32_t sentinel = parser.structural_indexes[parser.n_structural_indexes];
+  uint32_t write_idx = 0;
+  uint32_t last_rs_pos = 0;
+  uint32_t rs_count = 0;
+
+  for (uint32_t read_idx = 0; read_idx < parser.n_structural_indexes; read_idx++) {
+    const uint32_t pos = parser.structural_indexes[read_idx];
+    if (parser.buf[pos] == 0x1E) {
+      // This is an RS character - find the actual JSON value start.
+      last_rs_pos = pos;
+      rs_count++;
+      // Skip past this RS and any whitespace *and any additional RSes*
+      // to locate the real value. Consecutive RSes are degenerate
+      // "empty records" per RFC 7464; we collapse them here. They do
+      // not always appear as separate entries in structural_indexes
+      // because the scanner groups runs of adjacent non-whitespace
+      // scalar bytes (including RS) into a single scalar start.
+      uint32_t value_start = pos + 1;
+      while (value_start < scan_len) {
+        const uint8_t c = parser.buf[value_start];
+        if (c == ' ' || c == '\t' || c == '\n' || c == '\r') {
+          value_start++;
+        } else if (c == 0x1E) {
+          // Collapsed empty record. Still count it so rs_count reflects
+          // the true number of record markers and last_rs_pos tracks
+          // the final one.
+          last_rs_pos = value_start;
+          rs_count++;
+          value_start++;
+        } else {
+          break;
+        }
+      }
+      // If the scanner emitted additional structurals inside the
+      // whitespace+RS run we just walked over (i.e., isolated RSes
+      // separated by whitespace), skip past them so we do not
+      // double-count or double-emit.
+      while (read_idx + 1 < parser.n_structural_indexes &&
+             parser.structural_indexes[read_idx + 1] < value_start) {
+        read_idx++;
+      }
+      // Check if the value start is an operator (always present in
+      // scanner structural_indexes) or a scalar-like start (which may
+      // be missing from structural_indexes and must be added here).
+      // Note: '"' is NOT always in structural_indexes. The scanner
+      // classifies '"' as a scalar character and emits it as a
+      // structural only when it is a *scalar start* (preceded by
+      // whitespace or an operator). When '"' immediately follows an
+      // RS (which the scanner also classifies as scalar), it is
+      // treated as a scalar continuation and not emitted - so we
+      // must add it here just like any other scalar value.
+      if (value_start < scan_len) {
+        const uint8_t c = parser.buf[value_start];
+        const bool is_operator =
+            (c == '{' || c == '}' || c == '[' || c == ']' ||
+             c == ':' || c == ',');
+        // If the next scanner structural is exactly at value_start,
+        // the scanner already emitted it (it followed whitespace) and
+        // we must not add a duplicate - a subsequent iteration will
+        // copy it into write_idx.
+        const bool already_emitted =
+            (read_idx + 1 < parser.n_structural_indexes &&
+             parser.structural_indexes[read_idx + 1] == value_start);
+        if (!is_operator && !already_emitted) {
+          // Scalar value (number/true/false/null/string) - add its
+          // position since scanner missed it.
+          parser.structural_indexes[write_idx++] = value_start;
+        }
+      }
+    } else {
+      // Not RS, copy to output
+      parser.structural_indexes[write_idx++] = pos;
+    }
+  }
+
+  // Update structural index count. Compaction left a stale index in the slot
+  // past the end: restore the EOF sentinel that stage 1 had planted there, which
+  // document_stream::truncated_bytes() reads after a final batch.
+  parser.n_structural_indexes = write_idx;
+  parser.structural_indexes[write_idx] = sentinel;
+
+  if (parser.n_structural_indexes == 0) {
+    // Only RS markers here: the last one opens a record continuing past the
+    // window, so that is where the next batch resumes.
+    if (!is_final && rs_count > 0) { next_batch_start = last_rs_pos; }
+    return 0;
+  }
+  if (rs_count == 0) {
+    // No RS found; for final batch, try generic boundary detection
+    return is_final ? find_next_document_index(parser) : 0;
+  }
+
+  // Phase 2: Determine batch boundaries based on RS positions
+
+  if (is_final) {
+    // Final batch: all documents are complete (last one ends at EOF).
+    // In json_sequence mode, RS markers define document boundaries, so all
+    // remaining structurals form complete documents. Return them all directly.
+    // (Calling find_next_document_index() would fail for scalar documents.)
+    return parser.n_structural_indexes;
+  }
+
+  // Partial batch: need to find complete documents only.
+  // A document starting at an RS is complete if there is another RS after it.
+  next_batch_start = last_rs_pos;
+
+  if (rs_count < 2) {
+    // Only one RS, so we have at most one document that may be incomplete.
+    // We cannot confirm it is complete without another RS.
+    // Return DOCUMENT_TOO_LARGE if content was found (write_idx > 0), 0 if only separators.
+    return (parser.n_structural_indexes > 0) ? DOCUMENT_TOO_LARGE : 0;
+  }
+
+  // We have at least 2 RS markers. The last complete document ends before last_rs_pos.
+
+  // Find the structural index cutoff: keep only structurals < last_rs_pos.
+  // Since we already filtered RS, all remaining structurals are valid.
+  // We iterate backward to find the last structural before last_rs_pos.
+  uint32_t keep_count = 0;
+  for (uint32_t i = parser.n_structural_indexes; i > 0; i--) {
+    if (parser.structural_indexes[i - 1] < last_rs_pos) {
+      keep_count = i;
+      break;
+    }
+  }
+
+  // No structurals before the last RS - no complete documents
+  if (keep_count == 0) { return 0; }
+
+  // All documents before the last RS are complete by definition (the next RS
+  // confirms their end). No need to call find_next_document_index() which
+  // would fail for scalar documents like `1` or `"hello"`.
+  return keep_count;
+}
+
+/**
+ * Filter comma-delimited documents by removing root-level commas from
+ * structural indexes.
+ *
+ * For comma-delimited format like `{...},{...},{...}`, we need to remove
+ * the commas that separate documents (depth 0) while preserving commas
+ * inside arrays and objects (depth > 0).
+ *
+ * After filtering, the structural indexes look like whitespace-delimited
+ * documents, so find_next_document_index() works unchanged.
+ *
+ * @param parser The parser with structural_indexes and buf.
+ * @param len The length of the current batch buffer.
+ * @param is_final True if this is the final batch (no more data coming).
+ * @param next_batch_start Output: offset where the next batch should start.
+ * @return The number of structural indexes to keep,
+ *         0 if no document content found (EMPTY),
+ *         or DOCUMENT_TOO_LARGE if a document started but didn't fit (CAPACITY).
+ */
+simdjson_inline uint32_t filter_comma_delimited(
+    dom_parser_implementation &parser,
+    size_t len,
+    bool is_final,
+    uint32_t &next_batch_start) {
+  // Default: next batch starts at end of buffer
+  next_batch_start = uint32_t(len);
+
+  if (parser.n_structural_indexes == 0) { return 0; }
+
+  // Track depth to identify root-level commas (depth 0)
+  int depth = 0;
+  // The EOF sentinel: len, or where a discarded unclosed string starts.
+  const uint32_t sentinel = parser.structural_indexes[parser.n_structural_indexes];
+  uint32_t write_idx = 0;
+  uint32_t last_root_comma_pos = 0;
+  uint32_t root_comma_count = 0;
+
+  for (uint32_t i = 0; i < parser.n_structural_indexes; i++) {
+    uint32_t idx = parser.structural_indexes[i];
+    uint8_t c = parser.buf[idx];
+
+    switch (c) {
+      case '{': case '[':
+        depth++;
+        parser.structural_indexes[write_idx++] = idx;
+        break;
+      case '}': case ']':
+        depth--;
+        parser.structural_indexes[write_idx++] = idx;
+        break;
+      case ',':
+        if (depth == 0) {
+          // Root-level comma = document boundary, skip it
+          last_root_comma_pos = idx;
+          root_comma_count++;
+          continue;
+        }
+        parser.structural_indexes[write_idx++] = idx;
+        break;
+      default:
+        // Colons, scalars, etc.
+        parser.structural_indexes[write_idx++] = idx;
+        break;
+    }
+  }
+
+  // Update structural index count. Compaction left a stale index in the slot
+  // past the end: restore the EOF sentinel that stage 1 had planted there, which
+  // document_stream::truncated_bytes() reads after a final batch.
+  parser.n_structural_indexes = write_idx;
+  parser.structural_indexes[write_idx] = sentinel;
+
+  if (parser.n_structural_indexes == 0) { return 0; }
+
+  if (is_final) {
+    // Final batch: use standard boundary detection on filtered indexes
+    return find_next_document_index(parser);
+  }
+
+  // Partial batch: need to find complete documents only.
+  // A document ending with a root comma is complete.
+  if (root_comma_count == 0) {
+    // No root commas found; we cannot confirm any document is complete.
+    // The whole batch might be one incomplete document.
+    // Return DOCUMENT_TOO_LARGE if content was found (write_idx > 0), 0 if only commas.
+    return (parser.n_structural_indexes > 0) ? DOCUMENT_TOO_LARGE : 0;
+  }
+
+  // We have at least one root comma. Documents before the last comma are complete.
+  next_batch_start = last_root_comma_pos + 1;
+
+  // Find the structural index cutoff: keep only structurals < last_root_comma_pos
+  uint32_t keep_count = 0;
+  for (uint32_t i = parser.n_structural_indexes; i > 0; i--) {
+    if (parser.structural_indexes[i - 1] < last_root_comma_pos) {
+      keep_count = i;
+      break;
+    }
+  }
+
+  if (keep_count == 0) { return 0; }
+
+  // Use standard boundary detection on the complete portion
+  parser.n_structural_indexes = keep_count;
+  return find_next_document_index(parser);
+}
+
 } // namespace stage1
 } // unnamed namespace
 } // namespace rvv_vls
@@ -59804,7 +73909,6 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
         return EMPTY;
       }
     }
-
     parser.n_structural_indexes = new_structural_indexes;
   } else if (partial == stage1_mode::streaming_final) {
     if(have_unclosed_string) { parser.n_structural_indexes--; }
@@ -59832,6 +73936,75 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
         // the trailing garbage.
         return EMPTY;
     }
+  } else if (partial == stage1_mode::json_sequence_partial) {
+    // RFC 7464: use RS positions for batch boundaries
+    // A discarded unclosed string also caps how far the filter may scan.
+    size_t scan_len = len;
+    if(have_unclosed_string) {
+      parser.n_structural_indexes--;
+      if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return CAPACITY; }
+      scan_len = parser.structural_indexes[parser.n_structural_indexes];
+    }
+    uint32_t next_batch_start = uint32_t(len);
+    auto new_structural_indexes = find_next_document_index_json_sequence(parser, len, false, next_batch_start, scan_len);
+    if (new_structural_indexes == DOCUMENT_TOO_LARGE) {
+      return CAPACITY;
+    }
+    if (new_structural_indexes == 0) {
+      // An EMPTY batch must still advance next_batch_start, or document_stream
+      // re-parses the same bytes forever. CAPACITY when it cannot.
+      if (next_batch_start == 0) { return CAPACITY; }
+      parser.n_structural_indexes = 0;
+      parser.structural_indexes[0] = next_batch_start;
+      return EMPTY;
+    }
+    parser.n_structural_indexes = new_structural_indexes;
+    parser.structural_indexes[parser.n_structural_indexes] = next_batch_start;
+  } else if (partial == stage1_mode::json_sequence_final) {
+    // RFC 7464: final batch, last document extends to EOF
+    // As above: a discarded unclosed string caps how far the filter may scan.
+    size_t scan_len = len;
+    if(have_unclosed_string) {
+      parser.n_structural_indexes--;
+      scan_len = parser.structural_indexes[parser.n_structural_indexes];
+    }
+    uint32_t next_batch_start = uint32_t(len);
+    parser.n_structural_indexes = find_next_document_index_json_sequence(parser, len, true, next_batch_start, scan_len);
+    // The filter compacted structural_indexes in place and restored the EOF
+    // sentinel past the compacted end, so the copy below is either the start
+    // of a truncated document or len, as in streaming_final.
+    parser.structural_indexes[parser.n_structural_indexes + 1] = parser.structural_indexes[parser.n_structural_indexes];
+    parser.structural_indexes[parser.n_structural_indexes] = uint32_t(len);
+    if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return EMPTY; }
+  } else if (partial == stage1_mode::comma_delimited_partial) {
+    // Comma-delimited: filter root-level commas, use comma positions for batch boundaries
+    if(have_unclosed_string) {
+      parser.n_structural_indexes--;
+      if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return CAPACITY; }
+    }
+    uint32_t next_batch_start = uint32_t(len);
+    auto new_structural_indexes = filter_comma_delimited(parser, len, false, next_batch_start);
+    if (new_structural_indexes == DOCUMENT_TOO_LARGE) {
+      return CAPACITY;
+    }
+    if (new_structural_indexes == 0) {
+      // An EMPTY batch must still advance next_batch_start, or document_stream
+      // re-parses the same bytes forever. CAPACITY when it cannot.
+      if (next_batch_start == 0) { return CAPACITY; }
+      parser.n_structural_indexes = 0;
+      parser.structural_indexes[0] = next_batch_start;
+      return EMPTY;
+    }
+    parser.n_structural_indexes = new_structural_indexes;
+    parser.structural_indexes[parser.n_structural_indexes] = next_batch_start;
+  } else if (partial == stage1_mode::comma_delimited_final) {
+    // Comma-delimited: final batch, last document extends to EOF
+    if(have_unclosed_string) { parser.n_structural_indexes--; }
+    uint32_t next_batch_start = uint32_t(len);
+    parser.n_structural_indexes = filter_comma_delimited(parser, len, true, next_batch_start);
+    parser.structural_indexes[parser.n_structural_indexes + 1] = parser.structural_indexes[parser.n_structural_indexes];
+    parser.structural_indexes[parser.n_structural_indexes] = uint32_t(len);
+    if (simdjson_unlikely(parser.n_structural_indexes == 0u)) { return EMPTY; }
   }
   checker.check_eof();
   return checker.errors();
@@ -59914,7 +74087,6 @@ namespace {
 namespace stage2 {

 class json_iterator;
-class structural_iterator;
 struct tape_builder;
 struct tape_writer;

@@ -60211,7 +74383,7 @@ public:
    *
    * - increment_count(iter) - each time a value is found in an array or object.
    */
-  template<bool STREAMING, typename V>
+  template<bool STREAMING, bool UNPADDED, typename V>
   simdjson_warn_unused simdjson_inline error_code walk_document(V &visitor) noexcept;

   /**
@@ -60228,6 +74400,7 @@ public:
    *
    * They may include invalid JSON as well (such as `1.2.3` or `ture`).
    */
+  template<bool UNPADDED>
   simdjson_inline const uint8_t *peek() const noexcept;
   /**
    * Advance to the next token.
@@ -60236,6 +74409,7 @@ public:
    *
    * They may include invalid JSON as well (such as `1.2.3` or `ture`).
    */
+  template<bool UNPADDED>
   simdjson_inline const uint8_t *advance() noexcept;
   /**
    * Get the remaining length of the document, from the start of the current token.
@@ -60284,7 +74458,7 @@ public:
   simdjson_warn_unused simdjson_inline error_code visit_primitive(V &visitor, const uint8_t *value) noexcept;
 };

-template<bool STREAMING, typename V>
+template<bool STREAMING, bool UNPADDED, typename V>
 simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &visitor) noexcept {
   logger::log_start();

@@ -60299,7 +74473,7 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &
   // Read first value
   //
   {
-    auto value = advance();
+    auto value = advance<UNPADDED>();

     // Make sure the outer object or array is closed before continuing; otherwise, there are ways we
     // could get into memory corruption. See https://github.com/simdjson/simdjson/issues/906
@@ -60311,8 +74485,8 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &
     }

     switch (*value) {
-      case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
-      case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+      case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+      case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
       default: SIMDJSON_TRY( visitor.visit_root_primitive(*this, value) ); break;
     }
   }
@@ -60329,29 +74503,29 @@ object_begin:
   SIMDJSON_TRY( visitor.visit_object_start(*this) );

   {
-    auto key = advance();
+    auto key = advance<UNPADDED>();
     if (*key != '"') { log_error("Object does not start with a key"); return TAPE_ERROR; }
     SIMDJSON_TRY( visitor.increment_count(*this) );
     SIMDJSON_TRY( visitor.visit_key(*this, key) );
   }

 object_field:
-  if (simdjson_unlikely( *advance() != ':' )) { log_error("Missing colon after key in object"); return TAPE_ERROR; }
+  if (simdjson_unlikely( *advance<UNPADDED>() != ':' )) { log_error("Missing colon after key in object"); return TAPE_ERROR; }
   {
-    auto value = advance();
+    auto value = advance<UNPADDED>();
     switch (*value) {
-      case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
-      case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+      case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+      case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
       default: SIMDJSON_TRY( visitor.visit_primitive(*this, value) ); break;
     }
   }

 object_continue:
-  switch (*advance()) {
+  switch (*advance<UNPADDED>()) {
     case ',':
       SIMDJSON_TRY( visitor.increment_count(*this) );
       {
-        auto key = advance();
+        auto key = advance<UNPADDED>();
         if (simdjson_unlikely( *key != '"' )) { log_error("Key string missing at beginning of field in object"); return TAPE_ERROR; }
         SIMDJSON_TRY( visitor.visit_key(*this, key) );
       }
@@ -60379,16 +74553,16 @@ array_begin:

 array_value:
   {
-    auto value = advance();
+    auto value = advance<UNPADDED>();
     switch (*value) {
-      case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
-      case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+      case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+      case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
       default: SIMDJSON_TRY( visitor.visit_primitive(*this, value) ); break;
     }
   }

 array_continue:
-  switch (*advance()) {
+  switch (*advance<UNPADDED>()) {
     case ',': SIMDJSON_TRY( visitor.increment_count(*this) ); goto array_value;
     case ']': log_end_value("array"); SIMDJSON_TRY( visitor.visit_array_end(*this) ); goto scope_end;
     default: log_error("Missing comma between array values"); return TAPE_ERROR;
@@ -60416,11 +74590,29 @@ simdjson_inline json_iterator::json_iterator(dom_parser_implementation &_dom_par
     dom_parser{_dom_parser} {
 }

+// Stage 1 leaves a sentinel index of value `len`; reading buf[len] requires
+// padding. For UNPADDED, return a dummy so walk_document can fail cleanly (#2815).
+template<bool UNPADDED>
 simdjson_inline const uint8_t *json_iterator::peek() const noexcept {
-  return &buf[*(next_structural)];
+  const uint32_t idx = *(next_structural);
+  SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+    if (simdjson_unlikely(idx >= dom_parser.len)) {
+      static constexpr uint8_t unpadded_eof_sentinel = 0;
+      return &unpadded_eof_sentinel;
+    }
+  }
+  return &buf[idx];
 }
+template<bool UNPADDED>
 simdjson_inline const uint8_t *json_iterator::advance() noexcept {
-  return &buf[*(next_structural++)];
+  const uint32_t idx = *(next_structural++);
+  SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+    if (simdjson_unlikely(idx >= dom_parser.len)) {
+      static constexpr uint8_t unpadded_eof_sentinel = 0;
+      return &unpadded_eof_sentinel;
+    }
+  }
+  return &buf[idx];
 }
 simdjson_inline size_t json_iterator::remaining_len() const noexcept {
   return dom_parser.len - *(next_structural-1);
@@ -60460,7 +74652,20 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::visit_root_primit
     case '"': return visitor.visit_root_string(*this, value);
     case 't': return visitor.visit_root_true_atom(*this, value);
     case 'f': return visitor.visit_root_false_atom(*this, value);
+#if SIMDJSON_ENABLE_NAN_INF
+    case 'n': {
+      auto err = visitor.visit_root_null_atom(*this, value);
+      if (err == SUCCESS) { return err; }
+      // propagate the error value returned by a bad 'null' atom if parsing 'nan' fails
+      return visitor.visit_root_nan_atom(*this, value, err);
+    }
+    // 'N' isn't a canonically recognized atom, so we return a TAPE_ERROR if failure occurs
+    case 'N': return visitor.visit_root_nan_atom(*this, value, TAPE_ERROR);
+    case 'i':
+    case 'I': return visitor.visit_root_inf_atom(*this, value);
+#else
     case 'n': return visitor.visit_root_null_atom(*this, value);
+#endif
     case '-':
     case '0': case '1': case '2': case '3': case '4':
     case '5': case '6': case '7': case '8': case '9':
@@ -60482,7 +74687,20 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::visit_primitive(V
   switch (*value) {
     case 't': return visitor.visit_true_atom(*this, value);
     case 'f': return visitor.visit_false_atom(*this, value);
+#if SIMDJSON_ENABLE_NAN_INF
+    case 'n': {
+      auto err = visitor.visit_null_atom(*this, value);
+      if (err == SUCCESS) { return err; }
+      // propagate the error value returned by a bad 'null' atom if parsing 'nan' fails
+      return visitor.visit_nan_atom(*this, value, err);
+    }
+    // 'N' isn't a canonically recognized atom, so we return a TAPE_ERROR if failure occurs
+    case 'N': return visitor.visit_nan_atom(*this, value, TAPE_ERROR);
+    case 'i':
+    case 'I': return visitor.visit_inf_atom(*this, value);
+#else
     case 'n': return visitor.visit_null_atom(*this, value);
+#endif
     default:
       log_error("Non-value found when value was expected!");
       return TAPE_ERROR;
@@ -60668,9 +74886,13 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_string(const uint8_t *src, u
            within the unicode codepoint handling code. */
         src += bs_dist;
         dst += bs_dist;
-        if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
-          return nullptr;
-        }
+        // Decode adjacent Unicode escapes without returning to the
+        // quote-and-backslash scanner between code points.
+        do {
+          if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
+            return nullptr;
+          }
+        } while (src[0] == '\\' && src[1] == 'u');
       } else {
         /* simple 1:1 conversion. Will eat bs_dist+2 characters in input and
          * write bs_dist+1 characters to output
@@ -60693,6 +74915,77 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_string(const uint8_t *src, u
   }
 }

+/**
+ * Bounds-safe variant of parse_string for input buffers that are NOT padded to
+ * len + SIMDJSON_PADDING bytes. `buf_end` is one past the last readable input
+ * byte (buf + len). It behaves exactly like parse_string while we are at least
+ * SIMDJSON_PADDING bytes away from buf_end (so every speculative SIMD read stays
+ * in bounds); once within the final SIMDJSON_PADDING bytes it copies the few
+ * remaining bytes into a space-padded scratch buffer and finishes with the
+ * regular parse_string. This keeps the delicate escape/Unicode handling in one
+ * place (the proven parse_string) rather than duplicating it.
+ *
+ * Correctness relies on stage 1 having validated the string, i.e. there is an
+ * unescaped closing quote within [src, buf_end); that quote is therefore inside
+ * the copied scratch, so parse_string finds it without running off the scratch.
+ */
+simdjson_warn_unused simdjson_inline uint8_t *parse_string_safe(const uint8_t *src, uint8_t *dst, bool allow_replacement, const uint8_t *buf_end) {
+  // Far from the end: identical to parse_string's loop. The guard uses
+  // SIMDJSON_PADDING (>= BYTES_PROCESSED) so copy_and_find never reads past
+  // buf_end; escape/Unicode look-aheads read within the string (before the
+  // closing quote, which is < buf_end), so they are in bounds here too.
+  // We add margin (+12) for handle_unicode_codepoint's worst-case lookahead:
+  // after seeing a high surrogate, it does hex_to_u32_nocheck on the immediate
+  // following bytes (+6 from the '\'), then (if it sees \u) another
+  // hex_to_u32_nocheck at +8..+11 relative to the backslash that started the
+  // escape. With bs_dist up to BYTES_PROCESSED-1 this reaches +11 from the
+  // chunk start. The +12 margin ensures that even on kernels where
+  // BYTES_PROCESSED == SIMDJSON_PADDING (e.g. icelake) the 4-byte read stays
+  // in-bounds. The scratch fallback (3*PAD) is already safe.
+  while (src + SIMDJSON_PADDING + 12 <= buf_end) {
+    auto b = backslash_and_quote{};
+    auto bs_quote = b.copy_and_find(src, dst);
+    if (bs_quote.has_quote_first()) {
+      return dst + bs_quote.quote_index();
+    }
+    if (bs_quote.has_backslash()) {
+      auto bs_dist = bs_quote.backslash_index();
+      uint8_t escape_char = src[bs_dist + 1];
+      if (escape_char == 'u') {
+        src += bs_dist;
+        dst += bs_dist;
+        if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
+          return nullptr;
+        }
+      } else {
+        uint8_t escape_result = escape_map[escape_char];
+        if (escape_result == 0u) {
+          return nullptr;
+        }
+        dst[bs_dist] = escape_result;
+        src += bs_dist + 2;
+        dst += bs_dist + 1;
+      }
+    } else {
+      src += backslash_and_quote::BYTES_PROCESSED;
+      dst += backslash_and_quote::BYTES_PROCESSED;
+    }
+  }
+  // Within the final SIMDJSON_PADDING bytes: copy what remains into a
+  // space-padded scratch (spaces are neither quote nor backslash, so they do not
+  // disturb matching) and let the regular parser finish from there. The closing
+  // quote is within `remaining` (< SIMDJSON_PADDING), so parse_string finds it in
+  // the chunk starting at some offset <= remaining and reads at most
+  // BYTES_PROCESSED (<= SIMDJSON_PADDING) further -- i.e. under 2*SIMDJSON_PADDING.
+  // We size at 3x for a comfortable margin (the unicode look-ahead reads a few
+  // extra bytes past an escape).
+  uint8_t scratch[SIMDJSON_PADDING * 3];
+  const size_t remaining = size_t(buf_end - src); // < SIMDJSON_PADDING
+  std::memset(scratch, ' ', sizeof(scratch));
+  std::memcpy(scratch, src, remaining);
+  return parse_string(scratch, dst, allow_replacement);
+}
+
 simdjson_warn_unused simdjson_inline uint8_t *parse_wobbly_string(const uint8_t *src, uint8_t *dst) {
   // It is not ideal that this function is nearly identical to parse_string.
   while (1) {
@@ -60747,73 +75040,6 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_wobbly_string(const uint8_t

 #endif // SIMDJSON_SRC_GENERIC_STAGE2_STRINGPARSING_H
 /* end file generic/stage2/stringparsing.h for rvv_vls */
-/* including generic/stage2/structural_iterator.h for rvv_vls: #include <generic/stage2/structural_iterator.h> */
-/* begin file generic/stage2/structural_iterator.h for rvv_vls */
-#ifndef SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H
-
-/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
-/* amalgamation skipped (editor-only): #define SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H */
-/* amalgamation skipped (editor-only): #include <generic/stage2/base.h> */
-/* amalgamation skipped (editor-only): #include <simdjson/generic/dom_parser_implementation.h> */
-/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
-
-namespace simdjson {
-namespace rvv_vls {
-namespace {
-namespace stage2 {
-
-class structural_iterator {
-public:
-  const uint8_t* const buf;
-  uint32_t *next_structural;
-  dom_parser_implementation &dom_parser;
-
-  // Start a structural
-  simdjson_inline structural_iterator(dom_parser_implementation &_dom_parser, size_t start_structural_index)
-    : buf{_dom_parser.buf},
-      next_structural{&_dom_parser.structural_indexes[start_structural_index]},
-      dom_parser{_dom_parser} {
-  }
-  // Get the buffer position of the current structural character
-  simdjson_inline const uint8_t* current() {
-    return &buf[*(next_structural-1)];
-  }
-  // Get the current structural character
-  simdjson_inline char current_char() {
-    return buf[*(next_structural-1)];
-  }
-  // Get the next structural character without advancing
-  simdjson_inline char peek_next_char() {
-    return buf[*next_structural];
-  }
-  simdjson_inline const uint8_t* peek() {
-    return &buf[*next_structural];
-  }
-  simdjson_inline const uint8_t* advance() {
-    return &buf[*(next_structural++)];
-  }
-  simdjson_inline char advance_char() {
-    return buf[*(next_structural++)];
-  }
-  simdjson_inline size_t remaining_len() {
-    return dom_parser.len - *(next_structural-1);
-  }
-
-  simdjson_inline bool at_end() {
-    return next_structural == &dom_parser.structural_indexes[dom_parser.n_structural_indexes];
-  }
-  simdjson_inline bool at_beginning() {
-    return next_structural == dom_parser.structural_indexes.get();
-  }
-};
-
-} // namespace stage2
-} // unnamed namespace
-} // namespace rvv_vls
-} // namespace simdjson
-
-#endif // SIMDJSON_SRC_GENERIC_STAGE2_STRUCTURAL_ITERATOR_H
-/* end file generic/stage2/structural_iterator.h for rvv_vls */
 /* including generic/stage2/tape_builder.h for rvv_vls: #include <generic/stage2/tape_builder.h> */
 /* begin file generic/stage2/tape_builder.h for rvv_vls */
 #ifndef SIMDJSON_SRC_GENERIC_STAGE2_TAPE_BUILDER_H
@@ -60836,12 +75062,8 @@ namespace rvv_vls {
 namespace {
 namespace stage2 {

-struct tape_builder {
-  template<bool STREAMING>
-  simdjson_warn_unused static simdjson_inline error_code parse_document(
-    dom_parser_implementation &dom_parser,
-    dom::document &doc) noexcept;
-
+template <bool UNPADDED>
+struct tape_builder_impl {
   /** Called when a non-empty document starts. */
   simdjson_warn_unused simdjson_inline error_code visit_document_start(json_iterator &iter) noexcept;
   /** Called when a non-empty document ends without error. */
@@ -60894,88 +75116,130 @@ struct tape_builder {
   simdjson_warn_unused simdjson_inline error_code visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept;
   simdjson_warn_unused simdjson_inline error_code visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept;

+#if SIMDJSON_ENABLE_NAN_INF
+  simdjson_warn_unused simdjson_inline error_code visit_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept;
+  simdjson_warn_unused simdjson_inline error_code visit_root_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept;
+  // Attempts to parse 'inf' or 'infinity' (case insensitive). Because neither are canonical atoms,
+  // this returns a tape error on failure.
+  simdjson_warn_unused simdjson_inline error_code visit_inf_atom(json_iterator &iter, const uint8_t *value) noexcept;
+  simdjson_warn_unused simdjson_inline error_code visit_root_inf_atom(json_iterator &iter, const uint8_t *value) noexcept;
+#endif
+
   /** Called each time a new field or element in an array or object is found. */
   simdjson_warn_unused simdjson_inline error_code increment_count(json_iterator &iter) noexcept;

   /** Next location to write to tape */
   tape_writer tape;
+public:
+  simdjson_inline tape_builder_impl(dom::document &doc) noexcept;
 private:
   /** Next write location in the string buf for stage 2 parsing */
   uint8_t *current_string_buf_loc;

-  simdjson_inline tape_builder(dom::document &doc) noexcept;
-
   simdjson_inline uint32_t next_tape_index(json_iterator &iter) const noexcept;
   simdjson_inline void start_container(json_iterator &iter) noexcept;
   simdjson_warn_unused simdjson_inline error_code end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept;
   simdjson_warn_unused simdjson_inline error_code empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept;
   simdjson_inline uint8_t *on_start_string(json_iterator &iter) noexcept;
   simdjson_inline void on_end_string(uint8_t *dst) noexcept;
-}; // struct tape_builder
+}; // struct tape_builder_impl

-template<bool STREAMING>
-simdjson_warn_unused simdjson_inline error_code tape_builder::parse_document(
-    dom_parser_implementation &dom_parser,
-    dom::document &doc) noexcept {
-  dom_parser.doc = &doc;
-  json_iterator iter(dom_parser, STREAMING ? dom_parser.next_structural_index : 0);
-  tape_builder builder(doc);
-  return iter.walk_document<STREAMING>(builder);
-}
+// Thin, non-templated entry so each architecture's stage2() keeps calling
+// tape_builder::parse_document<STREAMING> unchanged. It chooses the bounds-safe
+// (unpadded) or the regular (padded) tape_builder_impl ONCE per document, so the
+// choice is a compile-time constant inside the walk: the padded path carries no
+// extra branch or load (see tape_builder_impl::visit_string).
+struct tape_builder {
+  template<bool STREAMING>
+  simdjson_warn_unused static simdjson_inline error_code parse_document(
+      dom_parser_implementation &dom_parser, dom::document &doc) noexcept {
+    dom_parser.doc = &doc;
+    json_iterator iter(dom_parser, STREAMING ? dom_parser.next_structural_index : 0);
+    if (dom_parser._unpadded) {
+      tape_builder_impl<true> builder(doc);
+      return iter.walk_document<STREAMING, true>(builder);
+    } else {
+      tape_builder_impl<false> builder(doc);
+      return iter.walk_document<STREAMING, false>(builder);
+    }
+  }
+};

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_primitive(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_primitive(json_iterator &iter, const uint8_t *value) noexcept {
   return iter.visit_root_primitive(*this, value);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_primitive(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_primitive(json_iterator &iter, const uint8_t *value) noexcept {
   return iter.visit_primitive(*this, value);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_empty_object(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_empty_object(json_iterator &iter) noexcept {
   return empty_container(iter, internal::tape_type::START_OBJECT, internal::tape_type::END_OBJECT);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_empty_array(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_empty_array(json_iterator &iter) noexcept {
   return empty_container(iter, internal::tape_type::START_ARRAY, internal::tape_type::END_ARRAY);
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_document_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_document_start(json_iterator &iter) noexcept {
   start_container(iter);
   return SUCCESS;
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_object_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_object_start(json_iterator &iter) noexcept {
   start_container(iter);
   return SUCCESS;
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_array_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_array_start(json_iterator &iter) noexcept {
   start_container(iter);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_object_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_object_end(json_iterator &iter) noexcept {
   return end_container(iter, internal::tape_type::START_OBJECT, internal::tape_type::END_OBJECT);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_array_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_array_end(json_iterator &iter) noexcept {
   return end_container(iter, internal::tape_type::START_ARRAY, internal::tape_type::END_ARRAY);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_document_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_document_end(json_iterator &iter) noexcept {
   constexpr uint32_t start_tape_index = 0;
   tape.append(start_tape_index, internal::tape_type::ROOT);
   tape_writer::write(iter.dom_parser.doc->tape[start_tape_index], next_tape_index(iter), internal::tape_type::ROOT);
   return SUCCESS;
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_key(json_iterator &iter, const uint8_t *key) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_key(json_iterator &iter, const uint8_t *key) noexcept {
   return visit_string(iter, key, true);
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::increment_count(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::increment_count(json_iterator &iter) noexcept {
   iter.dom_parser.open_containers[iter.depth].count++; // we have a key value pair in the object at parser.dom_parser.depth - 1
   return SUCCESS;
 }

-simdjson_inline tape_builder::tape_builder(dom::document &doc) noexcept : tape{doc.tape.get()}, current_string_buf_loc{doc.string_buf.get()} {}
+template <bool UNPADDED>
+simdjson_inline tape_builder_impl<UNPADDED>::tape_builder_impl(dom::document &doc) noexcept : tape{doc.tape.get()}, current_string_buf_loc{doc.string_buf.get()} {}

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_string(json_iterator &iter, const uint8_t *value, bool key) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_string(json_iterator &iter, const uint8_t *value, bool key) noexcept {
   iter.log_value(key ? "key" : "string");
   uint8_t *dst = on_start_string(iter);
-  dst = stringparsing::parse_string(value+1, dst, false); // We do not allow replacement when the escape characters are invalid.
+  // We do not allow replacement when the escape characters are invalid.
+  // UNPADDED is a compile-time constant chosen once per document by
+  // tape_builder::parse_document, so the padded build instantiates only the
+  // plain parse_string call below -- no runtime branch and no flag load.
+  SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+    dst = stringparsing::parse_string_safe(value+1, dst, false, iter.buf + iter.dom_parser.len);
+  } else {
+    dst = stringparsing::parse_string(value+1, dst, false);
+  }
   if (dst == nullptr) {
     iter.log_error("Invalid escape in string");
     return STRING_ERROR;
@@ -60984,27 +75248,48 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_string(json_
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_string(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_string(json_iterator &iter, const uint8_t *value) noexcept {
   return visit_string(iter, value);
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_number(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_number(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("number");
-  error_code err = numberparsing::parse_number(value, tape);
+  const uint8_t *num = value;
+  std::unique_ptr<uint8_t[]> copy{}; // keeps a padded copy of the tail alive when used
+  SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+    // numberparsing reads ahead in 8-byte blocks for floats
+    // (is_made_of_eight_digits_fast reads up to 7 bytes past the digits), so a
+    // number whose digits reach the final bytes of an unpadded buffer would read
+    // past it. *(next_structural) is the offset of the token following this
+    // number, hence an upper bound on where the digits end; when that is within
+    // SIMDJSON_PADDING of the end we parse from a space-padded copy of the tail
+    // (mirroring visit_root_number). This fires only for numbers near the end.
+    if (simdjson_unlikely(*(iter.next_structural) + SIMDJSON_PADDING > iter.dom_parser.len)) {
+      const size_t rl = iter.remaining_len(); // bytes from `value` to the end of the document
+      copy.reset(new (std::nothrow) uint8_t[rl + SIMDJSON_PADDING]);
+      if (copy.get() == nullptr) { return MEMALLOC; }
+      std::memcpy(copy.get(), value, rl);
+      std::memset(copy.get() + rl, ' ', SIMDJSON_PADDING);
+      num = copy.get();
+    }
+  }
+  error_code err = numberparsing::parse_number(num, tape);
   if (simdjson_unlikely(err == BIGINT_ERROR &&
       iter.dom_parser._number_as_string)) {
     // Write big integer to string buffer using the same format as strings.
     // Scan digits the same way parse_number does (skip optional '-', then digits).
-    const uint8_t *p = value;
+    const uint8_t *p = num;
     if (*p == '-') p++;
     while (numberparsing::is_digit(*p)) p++;
     // The digit run must be terminated by a structural or whitespace character; otherwise the
     // token is malformed (e.g. "123456789123456789123x").
     if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
-    size_t len = size_t(p - value);
+    size_t len = size_t(p - num);
     tape.append(current_string_buf_loc - iter.dom_parser.doc->string_buf.get(), internal::tape_type::BIGINT);
     uint8_t *dst = current_string_buf_loc + sizeof(uint32_t);
-    memcpy(dst, value, len);
+    memcpy(dst, num, len);
     dst += len;
     on_end_string(dst);
     return SUCCESS;
@@ -61012,7 +75297,8 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_number(json_
   return err;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_number(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_number(json_iterator &iter, const uint8_t *value) noexcept {
   //
   // We need to make a copy to make sure that the string is space terminated.
   // This is not about padding the input, which should already padded up
@@ -61026,76 +75312,142 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_number(
   // practice unless you are in the strange scenario where you have many JSON
   // documents made of single atoms.
   //
-  std::unique_ptr<uint8_t[]>copy(new (std::nothrow) uint8_t[iter.remaining_len() + SIMDJSON_PADDING]);
+  // In a stream, the input goes on with other documents: copy up to the next
+  // structural only, not to the end of the batch.
+  const size_t len = (std::min)(iter.remaining_len(), size_t(*iter.next_structural) - size_t(*(iter.next_structural - 1)));
+  std::unique_ptr<uint8_t[]>copy(new (std::nothrow) uint8_t[len + SIMDJSON_PADDING]);
   if (copy.get() == nullptr) { return MEMALLOC; }
-  std::memcpy(copy.get(), value, iter.remaining_len());
-  std::memset(copy.get() + iter.remaining_len(), ' ', SIMDJSON_PADDING);
+  std::memcpy(copy.get(), value, len);
+  std::memset(copy.get() + len, ' ', SIMDJSON_PADDING);
   error_code error = visit_number(iter, copy.get());
   return error;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("true");
-  if (!atomparsing::is_valid_true_atom(value)) { return T_ATOM_ERROR; }
+  // The non-length-aware validator reads a fixed 5 bytes; a malformed/truncated
+  // token at the very end of an unpadded buffer would over-read. Use the
+  // length-aware form there (the root variant already does this).
+  const bool ok = UNPADDED ? atomparsing::is_valid_true_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_true_atom(value);
+  if (!ok) { return T_ATOM_ERROR; }
   tape.append(0, internal::tape_type::TRUE_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("true");
   if (!atomparsing::is_valid_true_atom(value, iter.remaining_len())) { return T_ATOM_ERROR; }
   tape.append(0, internal::tape_type::TRUE_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("false");
-  if (!atomparsing::is_valid_false_atom(value)) { return F_ATOM_ERROR; }
+  const bool ok = UNPADDED ? atomparsing::is_valid_false_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_false_atom(value);
+  if (!ok) { return F_ATOM_ERROR; }
   tape.append(0, internal::tape_type::FALSE_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("false");
   if (!atomparsing::is_valid_false_atom(value, iter.remaining_len())) { return F_ATOM_ERROR; }
   tape.append(0, internal::tape_type::FALSE_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("null");
-  if (!atomparsing::is_valid_null_atom(value)) { return N_ATOM_ERROR; }
+  const bool ok = UNPADDED ? atomparsing::is_valid_null_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_null_atom(value);
+  if (!ok) { return N_ATOM_ERROR; }
   tape.append(0, internal::tape_type::NULL_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("null");
   if (!atomparsing::is_valid_null_atom(value, iter.remaining_len())) { return N_ATOM_ERROR; }
   tape.append(0, internal::tape_type::NULL_VALUE);
   return SUCCESS;
 }

+#if SIMDJSON_ENABLE_NAN_INF
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept {
+  iter.log_value("nan");
+  // For unpadded input use the length-aware validator so the 'infinity'-style
+  // 8-byte compare cannot read past the buffer on a malformed token at the end.
+  const bool ok = UNPADDED ? atomparsing::is_valid_nan_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_nan_atom(value);
+  if (!ok) { return errc; }
+  tape.append_double(std::numeric_limits<double>::quiet_NaN());
+  return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept {
+  iter.log_value("nan");
+  if (!atomparsing::is_valid_nan_atom(value, iter.remaining_len())) { return errc; }
+  tape.append_double(std::numeric_limits<double>::quiet_NaN());
+  return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_inf_atom(json_iterator &iter, const uint8_t *value) noexcept {
+  iter.log_value("inf");
+  // Because 'inf' is an extension, non a canonical atom, a tape error should be returned on failure.
+  // For unpadded input use the length-aware validator so the 'infinity' 8-byte
+  // compare cannot read past the buffer on a malformed token at the end.
+  const bool ok = UNPADDED ? atomparsing::is_valid_inf_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_inf_atom(value);
+  if (!ok) { return TAPE_ERROR; }
+  tape.append_double(std::numeric_limits<double>::infinity());
+  return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_inf_atom(json_iterator &iter, const uint8_t *value) noexcept {
+  iter.log_value("inf");
+  // Because 'inf' is an extension, non a canonical atom, a tape error should be returned on failure
+  if (!atomparsing::is_valid_inf_atom(value, iter.remaining_len())) { return TAPE_ERROR; }
+  tape.append_double(std::numeric_limits<double>::infinity());
+  return SUCCESS;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
 // private:

-simdjson_inline uint32_t tape_builder::next_tape_index(json_iterator &iter) const noexcept {
+template <bool UNPADDED>
+simdjson_inline uint32_t tape_builder_impl<UNPADDED>::next_tape_index(json_iterator &iter) const noexcept {
   return uint32_t(tape.next_tape_loc - iter.dom_parser.doc->tape.get());
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
   auto start_index = next_tape_index(iter);
   tape.append(start_index+2, start);
   tape.append(start_index, end);
   return SUCCESS;
 }

-simdjson_inline void tape_builder::start_container(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_inline void tape_builder_impl<UNPADDED>::start_container(json_iterator &iter) noexcept {
   iter.dom_parser.open_containers[iter.depth].tape_index = next_tape_index(iter);
   iter.dom_parser.open_containers[iter.depth].count = 0;
   tape.skip(); // We don't actually *write* the start element until the end.
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
   // Write the ending tape element, pointing at the start location
   const uint32_t start_tape_index = iter.dom_parser.open_containers[iter.depth].tape_index;
   tape.append(start_tape_index, end);
@@ -61108,13 +75460,15 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::end_container(json
   return SUCCESS;
 }

-simdjson_inline uint8_t *tape_builder::on_start_string(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_inline uint8_t *tape_builder_impl<UNPADDED>::on_start_string(json_iterator &iter) noexcept {
   // we advance the point, accounting for the fact that we have a NULL termination
   tape.append(current_string_buf_loc - iter.dom_parser.doc->string_buf.get(), internal::tape_type::STRING);
   return current_string_buf_loc + sizeof(uint32_t);
 }

-simdjson_inline void tape_builder::on_end_string(uint8_t *dst) noexcept {
+template <bool UNPADDED>
+simdjson_inline void tape_builder_impl<UNPADDED>::on_end_string(uint8_t *dst) noexcept {
   uint32_t str_length = uint32_t(dst - (current_string_buf_loc + sizeof(uint32_t)));
   // TODO check for overflow in case someone has a crazy string (>=4GB?)
   // But only add the overflow check when the document itself exceeds 4GB
@@ -61708,6 +76062,9 @@ namespace atomparsing {
 // to the compile-time constant 1936482662.
 simdjson_inline uint32_t string_to_uint32(const char* str) { uint32_t val; std::memcpy(&val, str, sizeof(uint32_t)); return val; }

+// Acts on the same principle as string_to_uint32, but on an 8-byte block of memory
+simdjson_inline uint64_t string_to_uint64(const char* str) { uint64_t val; std::memcpy(&val, str, sizeof(uint64_t)); return val; }
+

 // Again in str4ncmp we use a memcpy to avoid undefined behavior. The memcpy may appear expensive.
 // Yet all decent optimizing compilers will compile memcpy to a single instruction, just about.
@@ -61719,6 +76076,28 @@ simdjson_inline uint32_t str4ncmp(const uint8_t *src, const char* atom) {
   return srcval ^ string_to_uint32(atom);
 }

+// Checks that the first 8 characters of the input string match the given atom in a case-insensitive manner.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint64_t str8ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+  uint64_t srcval; // we want to avoid unaligned 32-bit loads (undefined in C/C++)
+  static_assert(sizeof(uint64_t) <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be larger than 8 bytes");
+  std::memcpy(&srcval, src, sizeof(uint64_t));
+
+  return (srcval | 0x2020202020202020ull) ^ string_to_uint64(atom);
+}
+
+// Checks that the first 3 characters of 'src' match 'atom' in a case-insensitive way.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint32_t str3ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+  return ((src[0] | 0x20) ^ atom[0]) //
+       | ((src[1] | 0x20) ^ atom[1]) //
+       | ((src[2] | 0x20) ^ atom[2]);
+}
+
 simdjson_warn_unused
 simdjson_inline bool is_valid_true_atom(const uint8_t *src) {
   return (str4ncmp(src, "true") | jsoncharutils::is_not_structural_or_whitespace(src[4])) == 0;
@@ -61755,6 +76134,71 @@ simdjson_inline bool is_valid_null_atom(const uint8_t *src, size_t len) {
   else { return false; }
 }

+#if SIMDJSON_ENABLE_NAN_INF
+// "nan" is 3 bytes; we check characters and then verify the next
+// character is structural or whitespace. We accept both "nan" and "NaN".
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src) {
+  return (str3ncmp_case_insensitive(src, "nan")
+        | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+}
+
+// checks that the next four characters of a string are 'nan"', where the 'nan'
+// is checked in a case-insensitive way.
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_in_string(const uint8_t *src) {
+  return (str3ncmp_case_insensitive(src, "nan") | (src[3] ^ '"')) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src, size_t len) {
+  if (len > 3) { return is_valid_nan_atom(src); }
+  if (len == 3) { return str3ncmp_case_insensitive(src, "nan") == 0; }
+  return false;
+}
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src) {
+  bool is_short_inf = (str3ncmp_case_insensitive(src, "inf")
+                     | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+  if(is_short_inf) return true;
+
+  // Check for 'infinity' (any capitalization)
+  return (str8ncmp_case_insensitive(src, "infinity") | jsoncharutils::is_not_structural_or_whitespace(src[8])) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_in_string(const uint8_t *src) {
+  bool is_short_inf = (str3ncmp_case_insensitive(src, "inf") | (src[3] ^ '"')) == 0;
+  if(is_short_inf) return true;
+
+  return (str8ncmp_case_insensitive(src, "infinity") | (src[8] ^ '"')) == 0;
+}
+
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src, size_t len) {
+  if (len > 8) { return is_valid_inf_atom(src); }
+  if (len == 8 && str8ncmp_case_insensitive(src, "infinity") == 0) {
+    return true;
+  }
+  if (len > 3) {
+    return (str3ncmp_case_insensitive(src, "inf")
+          | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+  }
+  if (len == 3) { return str3ncmp_case_insensitive(src, "inf") == 0; }
+  return false;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
 } // namespace atomparsing
 } // unnamed namespace
 } // namespace fallback
@@ -61845,7 +76289,7 @@ inline simdjson_warn_unused error_code dom_parser_implementation::set_capacity(s
 }

 inline simdjson_warn_unused error_code dom_parser_implementation::set_max_depth(size_t max_depth) noexcept {
-  if(max_depth == 0 || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
+  if(max_depth == 0 || max_depth > SIMDJSON_MAX_DEPTH || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
   // Stage 2 stacks
   open_containers.reset(new (std::nothrow) open_container[max_depth]);
   is_array.reset(new (std::nothrow) bool[max_depth]);
@@ -62024,6 +76468,7 @@ protected:
 /* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_NUMBERPARSING_H */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/base.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/jsoncharutils.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/atomparsing.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/internal/numberparsing_tables.h" */
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */

@@ -62063,6 +76508,18 @@ simdjson_inline double to_double(uint64_t mantissa, uint64_t real_exponent, bool
     return d;
 }

+// Convert a mantissa, an exponent and a sign bit into an ieee32 float (binary32).
+// The real_exponent needs to be in [0, 254] (technically real_exponent = 255 would be acceptable).
+// The mantissa should be in [0,1<<24). The bit at index (1U << 23) will be zeroed.
+simdjson_inline float to_float(uint32_t mantissa, uint32_t real_exponent, bool negative) {
+    float f;
+    mantissa &= ~(uint32_t(1) << 23);
+    mantissa |= real_exponent << 23;
+    mantissa |= ((static_cast<uint32_t>(negative)) << 31);
+    std::memcpy(&f, &mantissa, sizeof(f));
+    return f;
+}
+
 // Attempts to compute i * 10^(power) exactly; and if "negative" is
 // true, negate the result.
 // This function will only work in some cases, when it does not work, success is
@@ -62319,6 +76776,219 @@ simdjson_inline bool compute_float_64(int64_t power, uint64_t i, bool negative,
   return true;
 }

+// Attempts to compute i * 10^(power) as a binary32 (float) value; and if
+// "negative" is true, negate the result.
+//
+// This is the same Eisel-Lemire algorithm as compute_float_64, retargeted at
+// binary32. It is adapted from fast_float
+// (https://github.com/fastfloat/fast_float), where the routine is written once
+// and instantiated for each binary format. We only need 24 bits of mantissa
+// instead of 53, so we ask the truncated product for 24 + 2 = 26 accurate bits
+// (plus one bit that may be lost to the "upperbit" shift), and we shift the
+// product right by (upperbit + 64 - 23 - 3) instead of (upperbit + 64 - 52 - 3).
+//
+// The power-of-five table is shared with the binary64 code: the useful range
+// for binary32, [smallest_power_binary32, largest_power_binary32], is a strict
+// subset of [smallest_power, largest_power].
+//
+// The function returns false when the result would be infinite: simdjson
+// refuses to parse infinite values, so the caller reports an error. Unlike
+// compute_float_64, a false return never means "try harder": the accuracy of
+// the two-limb product is guaranteed by Noble Mushtak and Daniel Lemire,
+// "Fast Number Parsing Without Fallback" (https://arxiv.org/abs/2212.06644).
+// The caller must still fall back to a slow path when the 64-bit mantissa i
+// was truncated (more than 19 significant digits).
+//
+// We assume that power is in the [smallest_power, largest_power] interval: the
+// caller is responsible for this check.
+simdjson_inline bool compute_float_32(int64_t power, uint64_t i, bool negative, float &d) {
+  // Powers of ten that are exactly representable as binary32 values: 10^k is
+  // exact as long as k <= 10 (5^10 = 9765625 < 2**24 while 5^11 > 2**24).
+  static constexpr float power_of_ten_float[] = {
+      1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+      1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+  // The range of powers of ten that a non-zero, finite binary32 value can be
+  // built from. Anything smaller rounds to zero, anything larger is infinite.
+  // These are the smallest_power_of_ten()/largest_power_of_ten() constants that
+  // fast_float uses for binary32.
+  constexpr int smallest_power_binary32 = -65;
+  constexpr int largest_power_binary32 = 38;
+
+  // We start with the fast path described in
+  // Clinger WD. How to read floating point numbers accurately.
+  // ACM SIGPLAN Notices. 1990
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+  // We cannot be certain that x/y is rounded to nearest.
+  if (0 <= power && power <= 10 && i <= 16777215)
+#else
+  if (-10 <= power && power <= 10 && i <= 16777215)
+#endif
+  {
+    // Convert the integer into a float. This is lossless since
+    // 0 <= i <= 2^24 - 1.
+    d = float(i);
+    // Both d and the power of ten are exactly representable as binary32
+    // values, so the product (or quotient) is correctly rounded.
+    if (power < 0) {
+      d = d / power_of_ten_float[-power];
+    } else {
+      d = d * power_of_ten_float[power];
+    }
+    if (negative) {
+      d = -d;
+    }
+    return true;
+  }
+
+  // The fast path has failed, so we fall back on the Eisel-Lemire algorithm.
+  // It needs i > 0 (so that the leading bit of i can be normalized), so we
+  // handle i == 0 separately. We also handle the powers of ten that are so
+  // small (or so large) that the answer is zero (or infinite) whatever the
+  // mantissa is.
+  if (i == 0 || power < smallest_power_binary32) {
+    d = negative ? -0.0f : 0.0f;
+    return true;
+  }
+  if (power > largest_power_binary32) {
+    // We have, for sure, an infinite value.
+    return false;
+  }
+
+  // The exponent is 128 + 63 + power + floor(log(5**power)/log(2)).
+  // The 128 comes from the ieee32 standard (the minimal exponent is -127).
+  // The 63 comes from the fact that we use a 64-bit word.
+  // See compute_float_64 for a discussion of the magical 152170 + 65536.
+  int64_t exponent = (((152170 + 65536) * power) >> 16) + 128 + 63;
+
+  // We want the most significant bit of i to be 1. Shift if needed.
+  int lz = leading_zeroes(i);
+  i <<= lz;
+
+  // We want the most significant 64 bits of the product i * 5**power. It is
+  // safe to index the table because
+  // smallest_power <= smallest_power_binary32 <= power
+  //               <= largest_power_binary32 <= largest_power.
+  const uint32_t index = 2 * uint32_t(power - simdjson::internal::smallest_power);
+#if SIMDJSON_STATIC_REFLECTION
+  simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index]);
+#else
+  simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index]);
+#endif
+
+  // Unless the least significant 38 bits of the high (64-bit) part of the full
+  // product are all 1s, then we know that the most significant 26 bits are
+  // exact and no further work is needed. Having 26 bits is necessary because
+  // we need 24 bits for the mantissa but we have to have one rounding bit and
+  // we can waste a bit if the most significant bit of the product is zero.
+  // (For binary64, the same reasoning gives 55 bits and a 9-bit mask.)
+  if((firstproduct.high & 0x3FFFFFFFFF) == 0x3FFFFFFFFF) {
+    // The truncated multiplication was not accurate enough; use the next 64
+    // bits of the power of five to refine it. See compute_float_64 for a
+    // detailed discussion.
+#if SIMDJSON_STATIC_REFLECTION
+    simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index + 1]);
+#else
+    simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index + 1]);
+#endif
+    firstproduct.low += secondproduct.high;
+    if(secondproduct.high > firstproduct.low) { firstproduct.high++; }
+  }
+  uint64_t lower = firstproduct.low;
+  uint64_t upper = firstproduct.high;
+  // The final mantissa should be 24 bits with a leading 1.
+  // We shift it so that it occupies 25 bits with a leading 1.
+  ///////
+  uint64_t upperbit = upper >> 63;
+  uint64_t mantissa = upper >> (upperbit + 38); // 38 == 64 - 23 - 3
+  lz += int(1 ^ upperbit);
+
+  // Here we have mantissa < (1<<25).
+  int64_t real_exponent = exponent - lz;
+  if (simdjson_unlikely(real_exponent <= 0)) { // we have a subnormal?
+    // Here we have that real_exponent <= 0 so -real_exponent >= 0
+    if(-real_exponent + 1 >= 64) { // if we have more than 64 bits below the minimum exponent, you have a zero for sure.
+      d = negative ? -0.0f : 0.0f;
+      return true;
+    }
+    // next line is safe because -real_exponent + 1 < 64
+    mantissa >>= -real_exponent + 1;
+    // Thankfully, we can't have both "round-to-even" and subnormals because
+    // "round-to-even" only occurs for powers close to 0.
+    mantissa += (mantissa & 1); // round up
+    mantissa >>= 1;
+    // As in compute_float_64, rounding up may take us out of the subnormal
+    // range, so we can only decide after rounding.
+    real_exponent = (mantissa < (uint64_t(1) << 23)) ? 0 : 1;
+    d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+    return true;
+  }
+  // We have to round to even. The "to even" part is only a problem when we are
+  // right in between two floats, which we guard against. The bounds on the
+  // power of ten are those of fast_float's
+  // min_exponent_round_to_even()/max_exponent_round_to_even() for binary32:
+  // when q >= 0, (2m+1) must be divisible by 5^q with 5^q <= 2^25, so q <= 10;
+  // when q < 0, we need 2^24 x 5^{-q} < 2^{64}, so q >= -17.
+  if (simdjson_unlikely((lower <= 1) && (power >= -17) && (power <= 10) && ((mantissa & 3) == 1))) {
+    if((mantissa << (upperbit + 38)) == upper) {
+      mantissa &= ~uint64_t(1);   // flip it so that we do not round up
+    }
+  }
+
+  mantissa += mantissa & 1;
+  mantissa >>= 1;
+
+  // Here we have mantissa < (1<<24), unless there was an overflow
+  if (mantissa >= (uint64_t(1) << 24)) {
+    mantissa = (uint64_t(1) << 23);
+    real_exponent++;
+  }
+  mantissa &= ~(uint64_t(1) << 23);
+  // we have to check that real_exponent is in range, otherwise we bail out
+  if (simdjson_unlikely(real_exponent > 254)) {
+    // We have an infinite value!!! We could actually throw an error here if we could.
+    return false;
+  }
+  d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+  return true;
+}
+
+#if SIMDJSON_ENABLE_NAN_INF
+// Parses a nan or infinity. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, double& d) noexcept {
+  if (atomparsing::is_valid_inf_atom(src)) {
+    double inf = std::numeric_limits<double>::infinity();
+    d = negative ? -inf : inf;
+    return true;
+  }
+
+  if (atomparsing::is_valid_nan_atom(src)) {
+    d = std::numeric_limits<double>::quiet_NaN();
+    return true;
+  }
+
+  return false;
+}
+
+// Parses a nan or infinity as a binary32 value. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, float& d) noexcept {
+  if (atomparsing::is_valid_inf_atom(src)) {
+    float inf = std::numeric_limits<float>::infinity();
+    d = negative ? -inf : inf;
+    return true;
+  }
+
+  if (atomparsing::is_valid_nan_atom(src)) {
+    d = std::numeric_limits<float>::quiet_NaN();
+    return true;
+  }
+
+  return false;
+}
+#endif
+
 // We call a fallback floating-point parser that might be slow. Note
 // it will accept JSON numbers, but the JSON spec. is more restrictive so
 // before you call parse_float_fallback, you need to have validated the input
@@ -62356,6 +77026,16 @@ static bool parse_float_fallback(const uint8_t *ptr, const uint8_t *end_ptr, dou
   return !(*outDouble > (std::numeric_limits<double>::max)() || *outDouble < std::numeric_limits<double>::lowest());
 }

+// Same as parse_float_fallback, but for binary32 (float) values. Going through
+// the binary64 fallback and then rounding to binary32 would be subject to
+// double rounding, so we run the fallback algorithm directly on binary32.
+static bool parse_float_fallback(const uint8_t *ptr, float *outFloat) {
+  *outFloat = simdjson::internal::from_chars_float(reinterpret_cast<const char *>(ptr));
+  // We do not accept infinite values. See the binary64 version above for why we
+  // do not use std::isfinite.
+  return !(*outFloat > (std::numeric_limits<float>::max)() || *outFloat < std::numeric_limits<float>::lowest());
+}
+
 // check quickly whether the next 8 chars are made of digits
 // at a glance, it looks better than Mula's
 // http://0x80.pl/articles/swar-digits-validate.html
@@ -62374,6 +77054,28 @@ simdjson_inline bool is_made_of_eight_digits_fast(const uint8_t *chars) {
           0x3333333333333333);
 }

+// The same idea for four characters. A block of eight only bites when eight
+// digits are left, so a run of 4 to 7 digits used to fall back to one digit at a
+// time; taking four of them at once is what makes the tail of a long fraction
+// cheap. Adapted from fast_float, credit @aqrit.
+simdjson_inline bool is_made_of_four_digits_fast(const uint8_t *chars) {
+  uint32_t val;
+  // Reads up to 3 bytes beyond the digits, which SIMDJSON_PADDING covers.
+  static_assert(3 <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be bigger than 3");
+  std::memcpy(&val, chars, 4);
+  return !(((val + 0x46464646) | (val - 0x30303030)) & 0x80808080);
+}
+
+// Only call this when is_made_of_four_digits_fast() says the four characters are
+// digits. Little-endian: guarded by SIMDJSON_SWAR_NUMBER_PARSING.
+simdjson_inline uint32_t parse_four_digits_unrolled(const uint8_t *chars) {
+  uint32_t val;
+  std::memcpy(&val, chars, 4);
+  val -= 0x30303030;
+  val = (val * 10) + (val >> 8);
+  return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
 template<typename I>
 SIMDJSON_NO_SANITIZE_UNDEFINED // We deliberately allow overflow here and check later
 simdjson_inline bool parse_digit(const uint8_t c, I &i) {
@@ -62390,26 +77092,73 @@ simdjson_inline bool is_digit(const uint8_t c) {
   return static_cast<uint8_t>(c - '0') <= 9;
 }

-simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
-  // we continue with the fiction that we have an integer. If the
-  // floating point number is representable as x * 10^z for some integer
-  // z that fits in 53 bits, then we will be able to convert back the
-  // the integer into a float in a lossless manner.
-  const uint8_t *const first_after_period = p;
+// Consumes a run of digits into i, eight at a time while eight are available,
+// then four, then one at a time. Overflow is deliberate: the caller counts the
+// digits and falls back when there are too many for a 64-bit mantissa.
+//
+// Long runs of digits are what a float-heavy document is made of: the fraction
+// of a binary64 printed to full precision is 15 to 17 digits, and reading those
+// one at a time is the single largest cost in parse_double(). Integer parts are
+// left alone: they are usually a handful of digits, and there the check for a
+// block of eight is wasted work.
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_fraction_digits(const uint8_t *&p, uint64_t &i) {
+#ifdef SIMDJSON_SWAR_NUMBER_PARSING
+#if SIMDJSON_SWAR_NUMBER_PARSING
+  while (is_made_of_eight_digits_fast(p)) {
+    i = i * 100000000 + parse_eight_digits_unrolled(p);
+    p += 8;
+  }
+  // A 4 to 7 digit remainder is the common case once the blocks of eight are
+  // gone: 15 fraction digits leave 7, and 17 leave 1 after two blocks.
+  if (is_made_of_four_digits_fast(p)) {
+    i = i * 10000 + parse_four_digits_unrolled(p);
+    p += 4;
+  }
+#endif // SIMDJSON_SWAR_NUMBER_PARSING
+#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
+  while (parse_digit(*p, i)) { p++; }
+}

+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_float_integer_digits(const uint8_t *&p, uint64_t &i,
+                                                bool &leading_zero) {
+  if (!parse_digit(*p, i)) { leading_zero = true; return; }
+  p++;
+  leading_zero = (i == 0);
+  while (parse_digit(*p, i)) { p++; }
+}
+
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_integer_digits(const uint8_t *&p, uint64_t &i) {
 #ifdef SIMDJSON_SWAR_NUMBER_PARSING
 #if SIMDJSON_SWAR_NUMBER_PARSING
-  // this helps if we have lots of decimals!
-  // this turns out to be frequent enough.
+  // Identifiers, timestamps and counters often have eight digits or more.
+  // Prior related work: jsonifier parses integers as eight-digit SWAR words
+  // (str_to_i.hpp, https://github.com/nihilai-collective/Jsonifier). This
+  // takes one such word with parse_eight_digits_unrolled, the routine
+  // simdjson uses for long fractions.
   if (is_made_of_eight_digits_fast(p)) {
     i = i * 100000000 + parse_eight_digits_unrolled(p);
     p += 8;
   }
+  const uint8_t *const swar_end = p + 8;
+  while (p < swar_end && is_made_of_four_digits_fast(p)) {
+    i = i * 10000 + parse_four_digits_unrolled(p);
+    p += 4;
+  }
 #endif // SIMDJSON_SWAR_NUMBER_PARSING
 #endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
-  // Unrolling the first digit makes a small difference on some implementations (e.g. westmere)
-  if (parse_digit(*p, i)) { ++p; }
   while (parse_digit(*p, i)) { p++; }
+}
+
+simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
+  // we continue with the fiction that we have an integer. If the
+  // floating point number is representable as x * 10^z for some integer
+  // z that fits in 53 bits, then we will be able to convert back the
+  // the integer into a float in a lossless manner.
+  const uint8_t *const first_after_period = p;
+  parse_fraction_digits(p, i);
   exponent = first_after_period - p;
   // Decimal without digits (123.) is illegal
   if (exponent == 0) {
@@ -62498,7 +77247,7 @@ simdjson_inline size_t significant_digits(const uint8_t * start_digits, size_t d
 } // unnamed namespace

 /** @private */
-static error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
+inline error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
   if (parse_float_fallback(src, answer)) {
     return SUCCESS;
   }
@@ -62581,15 +77330,17 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
   return SUCCESS;              // always succeeds
 }

-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept  { return false; }
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept  { return false; }
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept { return number_type::signed_integer; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline bool is_negative(const uint8_t *) noexcept  { return false; }
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t *) noexcept  { return false; }
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t *) noexcept { return number_type::signed_integer; }
 #else

 // parse the number at src
@@ -62620,7 +77371,21 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
   size_t digit_count = size_t(p - start_digits);
-  if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) { return INVALID_NUMBER(src); }
+  if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // By this point, we know that our input does not begin with a digit. We will attempt
+    // to handle NaN/Infinity.
+
+    double d;
+    if (compute_nan_inf(p, negative, d)) {
+      writer.append_double(d);
+      return SUCCESS;
+    }
+#endif
+
+    return INVALID_NUMBER(src);
+  }

   //
   // Handle floats if there is a . or e (or both)
@@ -62669,7 +77434,7 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
     // - Therefore, if the number is positive and lower than that, it's overflow.
     // - The value we are looking at is less than or equal to INT64_MAX.
     //
-    }  else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return INVALID_NUMBER(src); }
+    }  else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return BIGINT_NUMBER(src); }
   }

   // Write unsigned if it does not fit in a signed integer.
@@ -62768,7 +77533,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const u
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -62866,7 +77631,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_stri
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -62921,7 +77686,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uin
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -63007,7 +77772,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = src;
   uint64_t i = 0;
-  while (parse_digit(*src, i)) { src++; }
+  parse_integer_digits(src, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -63047,9 +77812,247 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   //
   uint64_t i = 0;
   const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no loading digits, the number may be nan or infinity.
+    // Attempt to compute those, and return on success.
+    double d;
+    if (compute_nan_inf(p, negative, d)) { return d; }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = p-start_digits > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if (*p == 'e' || *p == 'E') {
+    p++;
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while (parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  double d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_64(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
+// Parse a JSON number into a binary32 (float) value.
+//
+// This mirrors parse_double, but it rounds to binary32 directly instead of
+// rounding to binary64 and then to binary32: the latter is subject to double
+// rounding and would not always produce the nearest float.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * src) noexcept {
+  //
+  // Check for minus sign
+  //
+  bool negative = (*src == '-');
+  src += uint8_t(negative);
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, the number may be nan or infinity.
+    // Attempt to compute those, and return on success.
+    float f;
+    if (compute_nan_inf(p, negative, f)) { return f; }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = p-start_digits > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if (*p == 'e' || *p == 'E') {
+    p++;
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while (parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  float d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_32(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
+simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
+  return (*src == '-');
+}
+
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept {
+  bool negative = (*src == '-');
+  src += uint8_t(negative);
+  const uint8_t *p = src;
+  while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
+  if ( p == src ) { return NUMBER_ERROR; }
+  if (jsoncharutils::is_structural_or_whitespace(*p)) { return true; }
+  return false;
+}
+
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept {
+  bool negative = (*src == '-');
+  src += uint8_t(negative);
+  const uint8_t *p = src;
+  while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
+  size_t digit_count = size_t(p - src);
+  if ( p == src ) { return NUMBER_ERROR; }
+  if (jsoncharutils::is_structural_or_whitespace(*p)) {
+    static const uint8_t * smaller_big_integer = reinterpret_cast<const uint8_t *>("9223372036854775808");
+    // We have an integer.
+    if(simdjson_unlikely(digit_count > 20)) {
+      return number_type::big_integer;
+    }
+    // If the number is negative and valid, it must be a signed integer.
+    if(negative) {
+      if (simdjson_unlikely(digit_count > 19)) return number_type::big_integer;
+      if (simdjson_unlikely(digit_count == 19 && memcmp(src, smaller_big_integer, 19) > 0)) {
+        return number_type::big_integer;
+      }
+#if SIMDJSON_MINUS_ZERO_AS_FLOAT
+      if(digit_count == 1 && src[0] == '0') {
+        // We have to write -0.0 instead of 0
+        return number_type::floating_point_number;
+      }
+#endif
+      return number_type::signed_integer;
+    }
+    // Let us check if we have a big integer (>=2**64).
+    static const uint8_t * two_to_sixtyfour = reinterpret_cast<const uint8_t *>("18446744073709551616");
+    if((digit_count > 20) || (digit_count == 20 && memcmp(src, two_to_sixtyfour, 20) >= 0)) {
+      return number_type::big_integer;
+    }
+    // The number is positive and smaller than 18446744073709551616 (or 2**64).
+    // We want values larger or equal to 9223372036854775808 to be unsigned
+    // integers, and the other values to be signed integers.
+    if((digit_count == 20) || (digit_count >= 19 && memcmp(src, smaller_big_integer, 19) >= 0)) {
+      return number_type::unsigned_integer;
+    }
+    return number_type::signed_integer;
+  }
+  // Hopefully, we have 'e' or 'E' or '.'.
+  return number_type::floating_point_number;
+}
+
+// Never read at src_end or beyond
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * src, const uint8_t * const src_end) noexcept {
+  if(src == src_end) { return NUMBER_ERROR; }
+  //
+  // Check for minus sign
+  //
+  bool negative = (*src == '-');
+  src += uint8_t(negative);
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  if(p == src_end) { return NUMBER_ERROR; }
   p += parse_digit(*p, i);
   bool leading_zero = (i == 0);
-  while (parse_digit(*p, i)) { p++; }
+  while ((p != src_end) && parse_digit(*p, i)) { p++; }
   // no integer digits, or 0123 (zero must be solo)
   if ( p == src ) { return INCORRECT_TYPE; }
   if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
@@ -63059,12 +78062,104 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   //
   int64_t exponent = 0;
   bool overflow;
-  if (simdjson_likely(*p == '.')) {
+  if (simdjson_likely((p != src_end) && (*p == '.'))) {
     p++;
     const uint8_t *start_decimal_digits = p;
-    if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
+    if ((p == src_end) || !parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
     p++;
-    while (parse_digit(*p, i)) { p++; }
+    while ((p != src_end) && parse_digit(*p, i)) { p++; }
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = start_digits-src > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if ((p != src_end) && (*p == 'e' || *p == 'E')) {
+    p++;
+    if(p == src_end) { return NUMBER_ERROR; }
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while ((p != src_end) && parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if ((p != src_end) && jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  double d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_64(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), src_end, &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * src) noexcept {
+  //
+  // Check for minus sign
+  //
+  bool negative = (*(src + 1) == '-');
+  src += uint8_t(negative) + 1;
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, attempt to parse numbers that are either
+    // NaN or Infinity
+    if (atomparsing::is_valid_inf_in_string(src)) {
+      double inf = std::numeric_limits<double>::infinity();
+      return negative ? -inf : inf;
+    }
+
+    if (atomparsing::is_valid_nan_in_string(src)) {
+      return std::numeric_limits<double>::quiet_NaN();
+    }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
     exponent = -(p - start_decimal_digits);

     // Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -63096,7 +78191,7 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
     exponent += exp_neg ? 0-exp : exp;
   }

-  if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+  if (*p != '"') { return NUMBER_ERROR; }

   overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;

@@ -63113,163 +78208,39 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   return d;
 }

-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
-  return (*src == '-');
-}
-
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept {
-  bool negative = (*src == '-');
-  src += uint8_t(negative);
-  const uint8_t *p = src;
-  while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
-  if ( p == src ) { return NUMBER_ERROR; }
-  if (jsoncharutils::is_structural_or_whitespace(*p)) { return true; }
-  return false;
-}
-
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept {
-  bool negative = (*src == '-');
-  src += uint8_t(negative);
-  const uint8_t *p = src;
-  while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
-  size_t digit_count = size_t(p - src);
-  if ( p == src ) { return NUMBER_ERROR; }
-  if (jsoncharutils::is_structural_or_whitespace(*p)) {
-    static const uint8_t * smaller_big_integer = reinterpret_cast<const uint8_t *>("9223372036854775808");
-    // We have an integer.
-    if(simdjson_unlikely(digit_count > 20)) {
-      return number_type::big_integer;
-    }
-    // If the number is negative and valid, it must be a signed integer.
-    if(negative) {
-      if (simdjson_unlikely(digit_count > 19)) return number_type::big_integer;
-      if (simdjson_unlikely(digit_count == 19 && memcmp(src, smaller_big_integer, 19) > 0)) {
-        return number_type::big_integer;
-      }
-#if SIMDJSON_MINUS_ZERO_AS_FLOAT
-      if(digit_count == 1 && src[0] == '0') {
-        // We have to write -0.0 instead of 0
-        return number_type::floating_point_number;
-      }
-#endif
-      return number_type::signed_integer;
-    }
-    // Let us check if we have a big integer (>=2**64).
-    static const uint8_t * two_to_sixtyfour = reinterpret_cast<const uint8_t *>("18446744073709551616");
-    if((digit_count > 20) || (digit_count == 20 && memcmp(src, two_to_sixtyfour, 20) >= 0)) {
-      return number_type::big_integer;
-    }
-    // The number is positive and smaller than 18446744073709551616 (or 2**64).
-    // We want values larger or equal to 9223372036854775808 to be unsigned
-    // integers, and the other values to be signed integers.
-    if((digit_count == 20) || (digit_count >= 19 && memcmp(src, smaller_big_integer, 19) >= 0)) {
-      return number_type::unsigned_integer;
-    }
-    return number_type::signed_integer;
-  }
-  // Hopefully, we have 'e' or 'E' or '.'.
-  return number_type::floating_point_number;
-}
-
-// Never read at src_end or beyond
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * src, const uint8_t * const src_end) noexcept {
-  if(src == src_end) { return NUMBER_ERROR; }
+// Parse a JSON number held inside a JSON string into a binary32 (float) value.
+// See parse_float for why we do not simply round parse_double_in_string.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * src) noexcept {
   //
   // Check for minus sign
   //
-  bool negative = (*src == '-');
-  src += uint8_t(negative);
+  bool negative = (*(src + 1) == '-');
+  src += uint8_t(negative) + 1;

   //
   // Parse the integer part.
   //
   uint64_t i = 0;
   const uint8_t *p = src;
-  if(p == src_end) { return NUMBER_ERROR; }
-  p += parse_digit(*p, i);
-  bool leading_zero = (i == 0);
-  while ((p != src_end) && parse_digit(*p, i)) { p++; }
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
   // no integer digits, or 0123 (zero must be solo)
-  if ( p == src ) { return INCORRECT_TYPE; }
-  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
-
-  //
-  // Parse the decimal part.
-  //
-  int64_t exponent = 0;
-  bool overflow;
-  if (simdjson_likely((p != src_end) && (*p == '.'))) {
-    p++;
-    const uint8_t *start_decimal_digits = p;
-    if ((p == src_end) || !parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
-    p++;
-    while ((p != src_end) && parse_digit(*p, i)) { p++; }
-    exponent = -(p - start_decimal_digits);
-
-    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
-    overflow = p-src-1 > 19;
-    if (simdjson_unlikely(overflow && leading_zero)) {
-      // Skip leading 0.00000 and see if it still overflows
-      const uint8_t *start_digits = src + 2;
-      while (*start_digits == '0') { start_digits++; }
-      overflow = start_digits-src > 19;
+  if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, attempt to parse numbers that are either
+    // NaN or Infinity
+    if (atomparsing::is_valid_inf_in_string(src)) {
+      float inf = std::numeric_limits<float>::infinity();
+      return negative ? -inf : inf;
     }
-  } else {
-    overflow = p-src > 19;
-  }
-
-  //
-  // Parse the exponent
-  //
-  if ((p != src_end) && (*p == 'e' || *p == 'E')) {
-    p++;
-    if(p == src_end) { return NUMBER_ERROR; }
-    bool exp_neg = *p == '-';
-    p += exp_neg || *p == '+';
-
-    uint64_t exp = 0;
-    const uint8_t *start_exp_digits = p;
-    while ((p != src_end) && parse_digit(*p, exp)) { p++; }
-    // no exp digits, or 20+ exp digits
-    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
-
-    exponent += exp_neg ? 0-exp : exp;
-  }
-
-  if ((p != src_end) && jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }

-  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+    if (atomparsing::is_valid_nan_in_string(src)) {
+      return std::numeric_limits<float>::quiet_NaN();
+    }
+#endif

-  //
-  // Assemble (or slow-parse) the float
-  //
-  double d;
-  if (simdjson_likely(!overflow)) {
-    if (compute_float_64(exponent, i, negative, d)) { return d; }
-  }
-  if (!parse_float_fallback(src - uint8_t(negative), src_end, &d)) {
-    return NUMBER_ERROR;
+    return INCORRECT_TYPE;
   }
-  return d;
-}
-
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * src) noexcept {
-  //
-  // Check for minus sign
-  //
-  bool negative = (*(src + 1) == '-');
-  src += uint8_t(negative) + 1;
-
-  //
-  // Parse the integer part.
-  //
-  uint64_t i = 0;
-  const uint8_t *p = src;
-  p += parse_digit(*p, i);
-  bool leading_zero = (i == 0);
-  while (parse_digit(*p, i)) { p++; }
-  // no integer digits, or 0123 (zero must be solo)
-  if ( p == src ) { return INCORRECT_TYPE; }
   if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }

   //
@@ -63280,9 +78251,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   if (simdjson_likely(*p == '.')) {
     p++;
     const uint8_t *start_decimal_digits = p;
-    if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
-    p++;
-    while (parse_digit(*p, i)) { p++; }
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
     exponent = -(p - start_decimal_digits);

     // Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -63321,9 +78291,9 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   //
   // Assemble (or slow-parse) the float
   //
-  double d;
+  float d;
   if (simdjson_likely(!overflow)) {
-    if (compute_float_64(exponent, i, negative, d)) { return d; }
+    if (compute_float_32(exponent, i, negative, d)) { return d; }
   }
   if (!parse_float_fallback(src - uint8_t(negative), &d)) {
     return NUMBER_ERROR;
@@ -63871,6 +78841,296 @@ simdjson_inline uint32_t find_next_document_index(dom_parser_implementation &par
   return 0;
 }

+/**
+ * Sentinel value returned to indicate a document started but didn't fit
+ * (CAPACITY error), as opposed to 0 which means no document content found
+ * (EMPTY).
+ */
+constexpr uint32_t DOCUMENT_TOO_LARGE = UINT32_MAX;
+
+/**
+ * For RFC 7464 JSON text sequences, filter RS from structural indexes and
+ * find batch boundaries.
+ *
+ * In JSON sequence mode, RS (0x1E) marks the start of each JSON text.
+ * RS bytes appear in structural_indexes as they are classified as scalars.
+ * This function:
+ * 1. Scans structural_indexes to find and count RS positions
+ * 2. Filters RS out of structural_indexes in-place
+ * 3. Determines batch boundaries based on RS positions
+ *
+ * @param parser The parser with structural_indexes and buf.
+ * @param len The length of the current batch buffer.
+ * @param is_final True if this is the final batch (no more data coming).
+ * @param next_batch_start Output: offset where the next batch should start.
+ * @param scan_len Offset past which no value start may be derived. Equals len
+ *        unless stage 1 dropped a trailing unclosed string, whose bytes it
+ *        never validated.
+ * @return The number of structural indexes to keep (after RS filtering),
+ *         0 if no document content found (EMPTY),
+ *         or DOCUMENT_TOO_LARGE if a document started but didn't fit (CAPACITY).
+ */
+simdjson_inline uint32_t find_next_document_index_json_sequence(
+    dom_parser_implementation &parser,
+    size_t len,
+    bool is_final,
+    uint32_t &next_batch_start,
+    size_t scan_len) {
+  // Default: next batch starts at end of buffer
+  next_batch_start = uint32_t(len);
+
+  if (parser.n_structural_indexes == 0) { return 0; }
+
+  // Phase 1: Scan structural_indexes to find RS positions and handle them.
+  // RS marks the start of a JSON text. For objects/arrays, the '{' or '[' after RS
+  // is already in structural_indexes (it's an operator). For scalars like numbers,
+  // the digit following RS is NOT in structural_indexes because the scanner sees
+  // RS as a scalar, making the digit a scalar continuation, not a start.
+  // We must: (1) remove RS from structural_indexes, and (2) for scalars, add the
+  // actual value start position.
+  // The EOF sentinel: len, or where a discarded unclosed string starts.
+  const uint32_t sentinel = parser.structural_indexes[parser.n_structural_indexes];
+  uint32_t write_idx = 0;
+  uint32_t last_rs_pos = 0;
+  uint32_t rs_count = 0;
+
+  for (uint32_t read_idx = 0; read_idx < parser.n_structural_indexes; read_idx++) {
+    const uint32_t pos = parser.structural_indexes[read_idx];
+    if (parser.buf[pos] == 0x1E) {
+      // This is an RS character - find the actual JSON value start.
+      last_rs_pos = pos;
+      rs_count++;
+      // Skip past this RS and any whitespace *and any additional RSes*
+      // to locate the real value. Consecutive RSes are degenerate
+      // "empty records" per RFC 7464; we collapse them here. They do
+      // not always appear as separate entries in structural_indexes
+      // because the scanner groups runs of adjacent non-whitespace
+      // scalar bytes (including RS) into a single scalar start.
+      uint32_t value_start = pos + 1;
+      while (value_start < scan_len) {
+        const uint8_t c = parser.buf[value_start];
+        if (c == ' ' || c == '\t' || c == '\n' || c == '\r') {
+          value_start++;
+        } else if (c == 0x1E) {
+          // Collapsed empty record. Still count it so rs_count reflects
+          // the true number of record markers and last_rs_pos tracks
+          // the final one.
+          last_rs_pos = value_start;
+          rs_count++;
+          value_start++;
+        } else {
+          break;
+        }
+      }
+      // If the scanner emitted additional structurals inside the
+      // whitespace+RS run we just walked over (i.e., isolated RSes
+      // separated by whitespace), skip past them so we do not
+      // double-count or double-emit.
+      while (read_idx + 1 < parser.n_structural_indexes &&
+             parser.structural_indexes[read_idx + 1] < value_start) {
+        read_idx++;
+      }
+      // Check if the value start is an operator (always present in
+      // scanner structural_indexes) or a scalar-like start (which may
+      // be missing from structural_indexes and must be added here).
+      // Note: '"' is NOT always in structural_indexes. The scanner
+      // classifies '"' as a scalar character and emits it as a
+      // structural only when it is a *scalar start* (preceded by
+      // whitespace or an operator). When '"' immediately follows an
+      // RS (which the scanner also classifies as scalar), it is
+      // treated as a scalar continuation and not emitted - so we
+      // must add it here just like any other scalar value.
+      if (value_start < scan_len) {
+        const uint8_t c = parser.buf[value_start];
+        const bool is_operator =
+            (c == '{' || c == '}' || c == '[' || c == ']' ||
+             c == ':' || c == ',');
+        // If the next scanner structural is exactly at value_start,
+        // the scanner already emitted it (it followed whitespace) and
+        // we must not add a duplicate - a subsequent iteration will
+        // copy it into write_idx.
+        const bool already_emitted =
+            (read_idx + 1 < parser.n_structural_indexes &&
+             parser.structural_indexes[read_idx + 1] == value_start);
+        if (!is_operator && !already_emitted) {
+          // Scalar value (number/true/false/null/string) - add its
+          // position since scanner missed it.
+          parser.structural_indexes[write_idx++] = value_start;
+        }
+      }
+    } else {
+      // Not RS, copy to output
+      parser.structural_indexes[write_idx++] = pos;
+    }
+  }
+
+  // Update structural index count. Compaction left a stale index in the slot
+  // past the end: restore the EOF sentinel that stage 1 had planted there, which
+  // document_stream::truncated_bytes() reads after a final batch.
+  parser.n_structural_indexes = write_idx;
+  parser.structural_indexes[write_idx] = sentinel;
+
+  if (parser.n_structural_indexes == 0) {
+    // Only RS markers here: the last one opens a record continuing past the
+    // window, so that is where the next batch resumes.
+    if (!is_final && rs_count > 0) { next_batch_start = last_rs_pos; }
+    return 0;
+  }
+  if (rs_count == 0) {
+    // No RS found; for final batch, try generic boundary detection
+    return is_final ? find_next_document_index(parser) : 0;
+  }
+
+  // Phase 2: Determine batch boundaries based on RS positions
+
+  if (is_final) {
+    // Final batch: all documents are complete (last one ends at EOF).
+    // In json_sequence mode, RS markers define document boundaries, so all
+    // remaining structurals form complete documents. Return them all directly.
+    // (Calling find_next_document_index() would fail for scalar documents.)
+    return parser.n_structural_indexes;
+  }
+
+  // Partial batch: need to find complete documents only.
+  // A document starting at an RS is complete if there is another RS after it.
+  next_batch_start = last_rs_pos;
+
+  if (rs_count < 2) {
+    // Only one RS, so we have at most one document that may be incomplete.
+    // We cannot confirm it is complete without another RS.
+    // Return DOCUMENT_TOO_LARGE if content was found (write_idx > 0), 0 if only separators.
+    return (parser.n_structural_indexes > 0) ? DOCUMENT_TOO_LARGE : 0;
+  }
+
+  // We have at least 2 RS markers. The last complete document ends before last_rs_pos.
+
+  // Find the structural index cutoff: keep only structurals < last_rs_pos.
+  // Since we already filtered RS, all remaining structurals are valid.
+  // We iterate backward to find the last structural before last_rs_pos.
+  uint32_t keep_count = 0;
+  for (uint32_t i = parser.n_structural_indexes; i > 0; i--) {
+    if (parser.structural_indexes[i - 1] < last_rs_pos) {
+      keep_count = i;
+      break;
+    }
+  }
+
+  // No structurals before the last RS - no complete documents
+  if (keep_count == 0) { return 0; }
+
+  // All documents before the last RS are complete by definition (the next RS
+  // confirms their end). No need to call find_next_document_index() which
+  // would fail for scalar documents like `1` or `"hello"`.
+  return keep_count;
+}
+
+/**
+ * Filter comma-delimited documents by removing root-level commas from
+ * structural indexes.
+ *
+ * For comma-delimited format like `{...},{...},{...}`, we need to remove
+ * the commas that separate documents (depth 0) while preserving commas
+ * inside arrays and objects (depth > 0).
+ *
+ * After filtering, the structural indexes look like whitespace-delimited
+ * documents, so find_next_document_index() works unchanged.
+ *
+ * @param parser The parser with structural_indexes and buf.
+ * @param len The length of the current batch buffer.
+ * @param is_final True if this is the final batch (no more data coming).
+ * @param next_batch_start Output: offset where the next batch should start.
+ * @return The number of structural indexes to keep,
+ *         0 if no document content found (EMPTY),
+ *         or DOCUMENT_TOO_LARGE if a document started but didn't fit (CAPACITY).
+ */
+simdjson_inline uint32_t filter_comma_delimited(
+    dom_parser_implementation &parser,
+    size_t len,
+    bool is_final,
+    uint32_t &next_batch_start) {
+  // Default: next batch starts at end of buffer
+  next_batch_start = uint32_t(len);
+
+  if (parser.n_structural_indexes == 0) { return 0; }
+
+  // Track depth to identify root-level commas (depth 0)
+  int depth = 0;
+  // The EOF sentinel: len, or where a discarded unclosed string starts.
+  const uint32_t sentinel = parser.structural_indexes[parser.n_structural_indexes];
+  uint32_t write_idx = 0;
+  uint32_t last_root_comma_pos = 0;
+  uint32_t root_comma_count = 0;
+
+  for (uint32_t i = 0; i < parser.n_structural_indexes; i++) {
+    uint32_t idx = parser.structural_indexes[i];
+    uint8_t c = parser.buf[idx];
+
+    switch (c) {
+      case '{': case '[':
+        depth++;
+        parser.structural_indexes[write_idx++] = idx;
+        break;
+      case '}': case ']':
+        depth--;
+        parser.structural_indexes[write_idx++] = idx;
+        break;
+      case ',':
+        if (depth == 0) {
+          // Root-level comma = document boundary, skip it
+          last_root_comma_pos = idx;
+          root_comma_count++;
+          continue;
+        }
+        parser.structural_indexes[write_idx++] = idx;
+        break;
+      default:
+        // Colons, scalars, etc.
+        parser.structural_indexes[write_idx++] = idx;
+        break;
+    }
+  }
+
+  // Update structural index count. Compaction left a stale index in the slot
+  // past the end: restore the EOF sentinel that stage 1 had planted there, which
+  // document_stream::truncated_bytes() reads after a final batch.
+  parser.n_structural_indexes = write_idx;
+  parser.structural_indexes[write_idx] = sentinel;
+
+  if (parser.n_structural_indexes == 0) { return 0; }
+
+  if (is_final) {
+    // Final batch: use standard boundary detection on filtered indexes
+    return find_next_document_index(parser);
+  }
+
+  // Partial batch: need to find complete documents only.
+  // A document ending with a root comma is complete.
+  if (root_comma_count == 0) {
+    // No root commas found; we cannot confirm any document is complete.
+    // The whole batch might be one incomplete document.
+    // Return DOCUMENT_TOO_LARGE if content was found (write_idx > 0), 0 if only commas.
+    return (parser.n_structural_indexes > 0) ? DOCUMENT_TOO_LARGE : 0;
+  }
+
+  // We have at least one root comma. Documents before the last comma are complete.
+  next_batch_start = last_root_comma_pos + 1;
+
+  // Find the structural index cutoff: keep only structurals < last_root_comma_pos
+  uint32_t keep_count = 0;
+  for (uint32_t i = parser.n_structural_indexes; i > 0; i--) {
+    if (parser.structural_indexes[i - 1] < last_root_comma_pos) {
+      keep_count = i;
+      break;
+    }
+  }
+
+  if (keep_count == 0) { return 0; }
+
+  // Use standard boundary detection on the complete portion
+  parser.n_structural_indexes = keep_count;
+  return find_next_document_index(parser);
+}
+
 } // namespace stage1
 } // unnamed namespace
 } // namespace fallback
@@ -64050,9 +79310,13 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_string(const uint8_t *src, u
            within the unicode codepoint handling code. */
         src += bs_dist;
         dst += bs_dist;
-        if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
-          return nullptr;
-        }
+        // Decode adjacent Unicode escapes without returning to the
+        // quote-and-backslash scanner between code points.
+        do {
+          if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
+            return nullptr;
+          }
+        } while (src[0] == '\\' && src[1] == 'u');
       } else {
         /* simple 1:1 conversion. Will eat bs_dist+2 characters in input and
          * write bs_dist+1 characters to output
@@ -64075,6 +79339,77 @@ simdjson_warn_unused simdjson_inline uint8_t *parse_string(const uint8_t *src, u
   }
 }

+/**
+ * Bounds-safe variant of parse_string for input buffers that are NOT padded to
+ * len + SIMDJSON_PADDING bytes. `buf_end` is one past the last readable input
+ * byte (buf + len). It behaves exactly like parse_string while we are at least
+ * SIMDJSON_PADDING bytes away from buf_end (so every speculative SIMD read stays
+ * in bounds); once within the final SIMDJSON_PADDING bytes it copies the few
+ * remaining bytes into a space-padded scratch buffer and finishes with the
+ * regular parse_string. This keeps the delicate escape/Unicode handling in one
+ * place (the proven parse_string) rather than duplicating it.
+ *
+ * Correctness relies on stage 1 having validated the string, i.e. there is an
+ * unescaped closing quote within [src, buf_end); that quote is therefore inside
+ * the copied scratch, so parse_string finds it without running off the scratch.
+ */
+simdjson_warn_unused simdjson_inline uint8_t *parse_string_safe(const uint8_t *src, uint8_t *dst, bool allow_replacement, const uint8_t *buf_end) {
+  // Far from the end: identical to parse_string's loop. The guard uses
+  // SIMDJSON_PADDING (>= BYTES_PROCESSED) so copy_and_find never reads past
+  // buf_end; escape/Unicode look-aheads read within the string (before the
+  // closing quote, which is < buf_end), so they are in bounds here too.
+  // We add margin (+12) for handle_unicode_codepoint's worst-case lookahead:
+  // after seeing a high surrogate, it does hex_to_u32_nocheck on the immediate
+  // following bytes (+6 from the '\'), then (if it sees \u) another
+  // hex_to_u32_nocheck at +8..+11 relative to the backslash that started the
+  // escape. With bs_dist up to BYTES_PROCESSED-1 this reaches +11 from the
+  // chunk start. The +12 margin ensures that even on kernels where
+  // BYTES_PROCESSED == SIMDJSON_PADDING (e.g. icelake) the 4-byte read stays
+  // in-bounds. The scratch fallback (3*PAD) is already safe.
+  while (src + SIMDJSON_PADDING + 12 <= buf_end) {
+    auto b = backslash_and_quote{};
+    auto bs_quote = b.copy_and_find(src, dst);
+    if (bs_quote.has_quote_first()) {
+      return dst + bs_quote.quote_index();
+    }
+    if (bs_quote.has_backslash()) {
+      auto bs_dist = bs_quote.backslash_index();
+      uint8_t escape_char = src[bs_dist + 1];
+      if (escape_char == 'u') {
+        src += bs_dist;
+        dst += bs_dist;
+        if (!handle_unicode_codepoint(&src, &dst, allow_replacement)) {
+          return nullptr;
+        }
+      } else {
+        uint8_t escape_result = escape_map[escape_char];
+        if (escape_result == 0u) {
+          return nullptr;
+        }
+        dst[bs_dist] = escape_result;
+        src += bs_dist + 2;
+        dst += bs_dist + 1;
+      }
+    } else {
+      src += backslash_and_quote::BYTES_PROCESSED;
+      dst += backslash_and_quote::BYTES_PROCESSED;
+    }
+  }
+  // Within the final SIMDJSON_PADDING bytes: copy what remains into a
+  // space-padded scratch (spaces are neither quote nor backslash, so they do not
+  // disturb matching) and let the regular parser finish from there. The closing
+  // quote is within `remaining` (< SIMDJSON_PADDING), so parse_string finds it in
+  // the chunk starting at some offset <= remaining and reads at most
+  // BYTES_PROCESSED (<= SIMDJSON_PADDING) further -- i.e. under 2*SIMDJSON_PADDING.
+  // We size at 3x for a comfortable margin (the unicode look-ahead reads a few
+  // extra bytes past an escape).
+  uint8_t scratch[SIMDJSON_PADDING * 3];
+  const size_t remaining = size_t(buf_end - src); // < SIMDJSON_PADDING
+  std::memset(scratch, ' ', sizeof(scratch));
+  std::memcpy(scratch, src, remaining);
+  return parse_string(scratch, dst, allow_replacement);
+}
+
 simdjson_warn_unused simdjson_inline uint8_t *parse_wobbly_string(const uint8_t *src, uint8_t *dst) {
   // It is not ideal that this function is nearly identical to parse_string.
   while (1) {
@@ -64278,7 +79613,7 @@ public:
    *
    * - increment_count(iter) - each time a value is found in an array or object.
    */
-  template<bool STREAMING, typename V>
+  template<bool STREAMING, bool UNPADDED, typename V>
   simdjson_warn_unused simdjson_inline error_code walk_document(V &visitor) noexcept;

   /**
@@ -64295,6 +79630,7 @@ public:
    *
    * They may include invalid JSON as well (such as `1.2.3` or `ture`).
    */
+  template<bool UNPADDED>
   simdjson_inline const uint8_t *peek() const noexcept;
   /**
    * Advance to the next token.
@@ -64303,6 +79639,7 @@ public:
    *
    * They may include invalid JSON as well (such as `1.2.3` or `ture`).
    */
+  template<bool UNPADDED>
   simdjson_inline const uint8_t *advance() noexcept;
   /**
    * Get the remaining length of the document, from the start of the current token.
@@ -64351,7 +79688,7 @@ public:
   simdjson_warn_unused simdjson_inline error_code visit_primitive(V &visitor, const uint8_t *value) noexcept;
 };

-template<bool STREAMING, typename V>
+template<bool STREAMING, bool UNPADDED, typename V>
 simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &visitor) noexcept {
   logger::log_start();

@@ -64366,7 +79703,7 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &
   // Read first value
   //
   {
-    auto value = advance();
+    auto value = advance<UNPADDED>();

     // Make sure the outer object or array is closed before continuing; otherwise, there are ways we
     // could get into memory corruption. See https://github.com/simdjson/simdjson/issues/906
@@ -64378,8 +79715,8 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::walk_document(V &
     }

     switch (*value) {
-      case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
-      case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+      case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+      case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
       default: SIMDJSON_TRY( visitor.visit_root_primitive(*this, value) ); break;
     }
   }
@@ -64396,29 +79733,29 @@ object_begin:
   SIMDJSON_TRY( visitor.visit_object_start(*this) );

   {
-    auto key = advance();
+    auto key = advance<UNPADDED>();
     if (*key != '"') { log_error("Object does not start with a key"); return TAPE_ERROR; }
     SIMDJSON_TRY( visitor.increment_count(*this) );
     SIMDJSON_TRY( visitor.visit_key(*this, key) );
   }

 object_field:
-  if (simdjson_unlikely( *advance() != ':' )) { log_error("Missing colon after key in object"); return TAPE_ERROR; }
+  if (simdjson_unlikely( *advance<UNPADDED>() != ':' )) { log_error("Missing colon after key in object"); return TAPE_ERROR; }
   {
-    auto value = advance();
+    auto value = advance<UNPADDED>();
     switch (*value) {
-      case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
-      case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+      case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+      case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
       default: SIMDJSON_TRY( visitor.visit_primitive(*this, value) ); break;
     }
   }

 object_continue:
-  switch (*advance()) {
+  switch (*advance<UNPADDED>()) {
     case ',':
       SIMDJSON_TRY( visitor.increment_count(*this) );
       {
-        auto key = advance();
+        auto key = advance<UNPADDED>();
         if (simdjson_unlikely( *key != '"' )) { log_error("Key string missing at beginning of field in object"); return TAPE_ERROR; }
         SIMDJSON_TRY( visitor.visit_key(*this, key) );
       }
@@ -64446,16 +79783,16 @@ array_begin:

 array_value:
   {
-    auto value = advance();
+    auto value = advance<UNPADDED>();
     switch (*value) {
-      case '{': if (*peek() == '}') { advance(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
-      case '[': if (*peek() == ']') { advance(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
+      case '{': if (*peek<UNPADDED>() == '}') { advance<UNPADDED>(); log_value("empty object"); SIMDJSON_TRY( visitor.visit_empty_object(*this) ); break; } goto object_begin;
+      case '[': if (*peek<UNPADDED>() == ']') { advance<UNPADDED>(); log_value("empty array"); SIMDJSON_TRY( visitor.visit_empty_array(*this) ); break; } goto array_begin;
       default: SIMDJSON_TRY( visitor.visit_primitive(*this, value) ); break;
     }
   }

 array_continue:
-  switch (*advance()) {
+  switch (*advance<UNPADDED>()) {
     case ',': SIMDJSON_TRY( visitor.increment_count(*this) ); goto array_value;
     case ']': log_end_value("array"); SIMDJSON_TRY( visitor.visit_array_end(*this) ); goto scope_end;
     default: log_error("Missing comma between array values"); return TAPE_ERROR;
@@ -64483,11 +79820,29 @@ simdjson_inline json_iterator::json_iterator(dom_parser_implementation &_dom_par
     dom_parser{_dom_parser} {
 }

+// Stage 1 leaves a sentinel index of value `len`; reading buf[len] requires
+// padding. For UNPADDED, return a dummy so walk_document can fail cleanly (#2815).
+template<bool UNPADDED>
 simdjson_inline const uint8_t *json_iterator::peek() const noexcept {
-  return &buf[*(next_structural)];
+  const uint32_t idx = *(next_structural);
+  SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+    if (simdjson_unlikely(idx >= dom_parser.len)) {
+      static constexpr uint8_t unpadded_eof_sentinel = 0;
+      return &unpadded_eof_sentinel;
+    }
+  }
+  return &buf[idx];
 }
+template<bool UNPADDED>
 simdjson_inline const uint8_t *json_iterator::advance() noexcept {
-  return &buf[*(next_structural++)];
+  const uint32_t idx = *(next_structural++);
+  SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+    if (simdjson_unlikely(idx >= dom_parser.len)) {
+      static constexpr uint8_t unpadded_eof_sentinel = 0;
+      return &unpadded_eof_sentinel;
+    }
+  }
+  return &buf[idx];
 }
 simdjson_inline size_t json_iterator::remaining_len() const noexcept {
   return dom_parser.len - *(next_structural-1);
@@ -64527,7 +79882,20 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::visit_root_primit
     case '"': return visitor.visit_root_string(*this, value);
     case 't': return visitor.visit_root_true_atom(*this, value);
     case 'f': return visitor.visit_root_false_atom(*this, value);
+#if SIMDJSON_ENABLE_NAN_INF
+    case 'n': {
+      auto err = visitor.visit_root_null_atom(*this, value);
+      if (err == SUCCESS) { return err; }
+      // propagate the error value returned by a bad 'null' atom if parsing 'nan' fails
+      return visitor.visit_root_nan_atom(*this, value, err);
+    }
+    // 'N' isn't a canonically recognized atom, so we return a TAPE_ERROR if failure occurs
+    case 'N': return visitor.visit_root_nan_atom(*this, value, TAPE_ERROR);
+    case 'i':
+    case 'I': return visitor.visit_root_inf_atom(*this, value);
+#else
     case 'n': return visitor.visit_root_null_atom(*this, value);
+#endif
     case '-':
     case '0': case '1': case '2': case '3': case '4':
     case '5': case '6': case '7': case '8': case '9':
@@ -64549,7 +79917,20 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::visit_primitive(V
   switch (*value) {
     case 't': return visitor.visit_true_atom(*this, value);
     case 'f': return visitor.visit_false_atom(*this, value);
+#if SIMDJSON_ENABLE_NAN_INF
+    case 'n': {
+      auto err = visitor.visit_null_atom(*this, value);
+      if (err == SUCCESS) { return err; }
+      // propagate the error value returned by a bad 'null' atom if parsing 'nan' fails
+      return visitor.visit_nan_atom(*this, value, err);
+    }
+    // 'N' isn't a canonically recognized atom, so we return a TAPE_ERROR if failure occurs
+    case 'N': return visitor.visit_nan_atom(*this, value, TAPE_ERROR);
+    case 'i':
+    case 'I': return visitor.visit_inf_atom(*this, value);
+#else
     case 'n': return visitor.visit_null_atom(*this, value);
+#endif
     default:
       log_error("Non-value found when value was expected!");
       return TAPE_ERROR;
@@ -64720,12 +80101,8 @@ namespace fallback {
 namespace {
 namespace stage2 {

-struct tape_builder {
-  template<bool STREAMING>
-  simdjson_warn_unused static simdjson_inline error_code parse_document(
-    dom_parser_implementation &dom_parser,
-    dom::document &doc) noexcept;
-
+template <bool UNPADDED>
+struct tape_builder_impl {
   /** Called when a non-empty document starts. */
   simdjson_warn_unused simdjson_inline error_code visit_document_start(json_iterator &iter) noexcept;
   /** Called when a non-empty document ends without error. */
@@ -64778,88 +80155,130 @@ struct tape_builder {
   simdjson_warn_unused simdjson_inline error_code visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept;
   simdjson_warn_unused simdjson_inline error_code visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept;

+#if SIMDJSON_ENABLE_NAN_INF
+  simdjson_warn_unused simdjson_inline error_code visit_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept;
+  simdjson_warn_unused simdjson_inline error_code visit_root_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept;
+  // Attempts to parse 'inf' or 'infinity' (case insensitive). Because neither are canonical atoms,
+  // this returns a tape error on failure.
+  simdjson_warn_unused simdjson_inline error_code visit_inf_atom(json_iterator &iter, const uint8_t *value) noexcept;
+  simdjson_warn_unused simdjson_inline error_code visit_root_inf_atom(json_iterator &iter, const uint8_t *value) noexcept;
+#endif
+
   /** Called each time a new field or element in an array or object is found. */
   simdjson_warn_unused simdjson_inline error_code increment_count(json_iterator &iter) noexcept;

   /** Next location to write to tape */
   tape_writer tape;
+public:
+  simdjson_inline tape_builder_impl(dom::document &doc) noexcept;
 private:
   /** Next write location in the string buf for stage 2 parsing */
   uint8_t *current_string_buf_loc;

-  simdjson_inline tape_builder(dom::document &doc) noexcept;
-
   simdjson_inline uint32_t next_tape_index(json_iterator &iter) const noexcept;
   simdjson_inline void start_container(json_iterator &iter) noexcept;
   simdjson_warn_unused simdjson_inline error_code end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept;
   simdjson_warn_unused simdjson_inline error_code empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept;
   simdjson_inline uint8_t *on_start_string(json_iterator &iter) noexcept;
   simdjson_inline void on_end_string(uint8_t *dst) noexcept;
-}; // struct tape_builder
+}; // struct tape_builder_impl

-template<bool STREAMING>
-simdjson_warn_unused simdjson_inline error_code tape_builder::parse_document(
-    dom_parser_implementation &dom_parser,
-    dom::document &doc) noexcept {
-  dom_parser.doc = &doc;
-  json_iterator iter(dom_parser, STREAMING ? dom_parser.next_structural_index : 0);
-  tape_builder builder(doc);
-  return iter.walk_document<STREAMING>(builder);
-}
+// Thin, non-templated entry so each architecture's stage2() keeps calling
+// tape_builder::parse_document<STREAMING> unchanged. It chooses the bounds-safe
+// (unpadded) or the regular (padded) tape_builder_impl ONCE per document, so the
+// choice is a compile-time constant inside the walk: the padded path carries no
+// extra branch or load (see tape_builder_impl::visit_string).
+struct tape_builder {
+  template<bool STREAMING>
+  simdjson_warn_unused static simdjson_inline error_code parse_document(
+      dom_parser_implementation &dom_parser, dom::document &doc) noexcept {
+    dom_parser.doc = &doc;
+    json_iterator iter(dom_parser, STREAMING ? dom_parser.next_structural_index : 0);
+    if (dom_parser._unpadded) {
+      tape_builder_impl<true> builder(doc);
+      return iter.walk_document<STREAMING, true>(builder);
+    } else {
+      tape_builder_impl<false> builder(doc);
+      return iter.walk_document<STREAMING, false>(builder);
+    }
+  }
+};

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_primitive(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_primitive(json_iterator &iter, const uint8_t *value) noexcept {
   return iter.visit_root_primitive(*this, value);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_primitive(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_primitive(json_iterator &iter, const uint8_t *value) noexcept {
   return iter.visit_primitive(*this, value);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_empty_object(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_empty_object(json_iterator &iter) noexcept {
   return empty_container(iter, internal::tape_type::START_OBJECT, internal::tape_type::END_OBJECT);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_empty_array(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_empty_array(json_iterator &iter) noexcept {
   return empty_container(iter, internal::tape_type::START_ARRAY, internal::tape_type::END_ARRAY);
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_document_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_document_start(json_iterator &iter) noexcept {
   start_container(iter);
   return SUCCESS;
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_object_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_object_start(json_iterator &iter) noexcept {
   start_container(iter);
   return SUCCESS;
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_array_start(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_array_start(json_iterator &iter) noexcept {
   start_container(iter);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_object_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_object_end(json_iterator &iter) noexcept {
   return end_container(iter, internal::tape_type::START_OBJECT, internal::tape_type::END_OBJECT);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_array_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_array_end(json_iterator &iter) noexcept {
   return end_container(iter, internal::tape_type::START_ARRAY, internal::tape_type::END_ARRAY);
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_document_end(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_document_end(json_iterator &iter) noexcept {
   constexpr uint32_t start_tape_index = 0;
   tape.append(start_tape_index, internal::tape_type::ROOT);
   tape_writer::write(iter.dom_parser.doc->tape[start_tape_index], next_tape_index(iter), internal::tape_type::ROOT);
   return SUCCESS;
 }
-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_key(json_iterator &iter, const uint8_t *key) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_key(json_iterator &iter, const uint8_t *key) noexcept {
   return visit_string(iter, key, true);
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::increment_count(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::increment_count(json_iterator &iter) noexcept {
   iter.dom_parser.open_containers[iter.depth].count++; // we have a key value pair in the object at parser.dom_parser.depth - 1
   return SUCCESS;
 }

-simdjson_inline tape_builder::tape_builder(dom::document &doc) noexcept : tape{doc.tape.get()}, current_string_buf_loc{doc.string_buf.get()} {}
+template <bool UNPADDED>
+simdjson_inline tape_builder_impl<UNPADDED>::tape_builder_impl(dom::document &doc) noexcept : tape{doc.tape.get()}, current_string_buf_loc{doc.string_buf.get()} {}

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_string(json_iterator &iter, const uint8_t *value, bool key) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_string(json_iterator &iter, const uint8_t *value, bool key) noexcept {
   iter.log_value(key ? "key" : "string");
   uint8_t *dst = on_start_string(iter);
-  dst = stringparsing::parse_string(value+1, dst, false); // We do not allow replacement when the escape characters are invalid.
+  // We do not allow replacement when the escape characters are invalid.
+  // UNPADDED is a compile-time constant chosen once per document by
+  // tape_builder::parse_document, so the padded build instantiates only the
+  // plain parse_string call below -- no runtime branch and no flag load.
+  SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+    dst = stringparsing::parse_string_safe(value+1, dst, false, iter.buf + iter.dom_parser.len);
+  } else {
+    dst = stringparsing::parse_string(value+1, dst, false);
+  }
   if (dst == nullptr) {
     iter.log_error("Invalid escape in string");
     return STRING_ERROR;
@@ -64868,27 +80287,48 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_string(json_
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_string(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_string(json_iterator &iter, const uint8_t *value) noexcept {
   return visit_string(iter, value);
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_number(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_number(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("number");
-  error_code err = numberparsing::parse_number(value, tape);
+  const uint8_t *num = value;
+  std::unique_ptr<uint8_t[]> copy{}; // keeps a padded copy of the tail alive when used
+  SIMDJSON_IF_CONSTEXPR (UNPADDED) {
+    // numberparsing reads ahead in 8-byte blocks for floats
+    // (is_made_of_eight_digits_fast reads up to 7 bytes past the digits), so a
+    // number whose digits reach the final bytes of an unpadded buffer would read
+    // past it. *(next_structural) is the offset of the token following this
+    // number, hence an upper bound on where the digits end; when that is within
+    // SIMDJSON_PADDING of the end we parse from a space-padded copy of the tail
+    // (mirroring visit_root_number). This fires only for numbers near the end.
+    if (simdjson_unlikely(*(iter.next_structural) + SIMDJSON_PADDING > iter.dom_parser.len)) {
+      const size_t rl = iter.remaining_len(); // bytes from `value` to the end of the document
+      copy.reset(new (std::nothrow) uint8_t[rl + SIMDJSON_PADDING]);
+      if (copy.get() == nullptr) { return MEMALLOC; }
+      std::memcpy(copy.get(), value, rl);
+      std::memset(copy.get() + rl, ' ', SIMDJSON_PADDING);
+      num = copy.get();
+    }
+  }
+  error_code err = numberparsing::parse_number(num, tape);
   if (simdjson_unlikely(err == BIGINT_ERROR &&
       iter.dom_parser._number_as_string)) {
     // Write big integer to string buffer using the same format as strings.
     // Scan digits the same way parse_number does (skip optional '-', then digits).
-    const uint8_t *p = value;
+    const uint8_t *p = num;
     if (*p == '-') p++;
     while (numberparsing::is_digit(*p)) p++;
     // The digit run must be terminated by a structural or whitespace character; otherwise the
     // token is malformed (e.g. "123456789123456789123x").
     if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
-    size_t len = size_t(p - value);
+    size_t len = size_t(p - num);
     tape.append(current_string_buf_loc - iter.dom_parser.doc->string_buf.get(), internal::tape_type::BIGINT);
     uint8_t *dst = current_string_buf_loc + sizeof(uint32_t);
-    memcpy(dst, value, len);
+    memcpy(dst, num, len);
     dst += len;
     on_end_string(dst);
     return SUCCESS;
@@ -64896,7 +80336,8 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_number(json_
   return err;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_number(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_number(json_iterator &iter, const uint8_t *value) noexcept {
   //
   // We need to make a copy to make sure that the string is space terminated.
   // This is not about padding the input, which should already padded up
@@ -64910,76 +80351,142 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_number(
   // practice unless you are in the strange scenario where you have many JSON
   // documents made of single atoms.
   //
-  std::unique_ptr<uint8_t[]>copy(new (std::nothrow) uint8_t[iter.remaining_len() + SIMDJSON_PADDING]);
+  // In a stream, the input goes on with other documents: copy up to the next
+  // structural only, not to the end of the batch.
+  const size_t len = (std::min)(iter.remaining_len(), size_t(*iter.next_structural) - size_t(*(iter.next_structural - 1)));
+  std::unique_ptr<uint8_t[]>copy(new (std::nothrow) uint8_t[len + SIMDJSON_PADDING]);
   if (copy.get() == nullptr) { return MEMALLOC; }
-  std::memcpy(copy.get(), value, iter.remaining_len());
-  std::memset(copy.get() + iter.remaining_len(), ' ', SIMDJSON_PADDING);
+  std::memcpy(copy.get(), value, len);
+  std::memset(copy.get() + len, ' ', SIMDJSON_PADDING);
   error_code error = visit_number(iter, copy.get());
   return error;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("true");
-  if (!atomparsing::is_valid_true_atom(value)) { return T_ATOM_ERROR; }
+  // The non-length-aware validator reads a fixed 5 bytes; a malformed/truncated
+  // token at the very end of an unpadded buffer would over-read. Use the
+  // length-aware form there (the root variant already does this).
+  const bool ok = UNPADDED ? atomparsing::is_valid_true_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_true_atom(value);
+  if (!ok) { return T_ATOM_ERROR; }
   tape.append(0, internal::tape_type::TRUE_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_true_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("true");
   if (!atomparsing::is_valid_true_atom(value, iter.remaining_len())) { return T_ATOM_ERROR; }
   tape.append(0, internal::tape_type::TRUE_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("false");
-  if (!atomparsing::is_valid_false_atom(value)) { return F_ATOM_ERROR; }
+  const bool ok = UNPADDED ? atomparsing::is_valid_false_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_false_atom(value);
+  if (!ok) { return F_ATOM_ERROR; }
   tape.append(0, internal::tape_type::FALSE_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_false_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("false");
   if (!atomparsing::is_valid_false_atom(value, iter.remaining_len())) { return F_ATOM_ERROR; }
   tape.append(0, internal::tape_type::FALSE_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("null");
-  if (!atomparsing::is_valid_null_atom(value)) { return N_ATOM_ERROR; }
+  const bool ok = UNPADDED ? atomparsing::is_valid_null_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_null_atom(value);
+  if (!ok) { return N_ATOM_ERROR; }
   tape.append(0, internal::tape_type::NULL_VALUE);
   return SUCCESS;
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_null_atom(json_iterator &iter, const uint8_t *value) noexcept {
   iter.log_value("null");
   if (!atomparsing::is_valid_null_atom(value, iter.remaining_len())) { return N_ATOM_ERROR; }
   tape.append(0, internal::tape_type::NULL_VALUE);
   return SUCCESS;
 }

+#if SIMDJSON_ENABLE_NAN_INF
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept {
+  iter.log_value("nan");
+  // For unpadded input use the length-aware validator so the 'infinity'-style
+  // 8-byte compare cannot read past the buffer on a malformed token at the end.
+  const bool ok = UNPADDED ? atomparsing::is_valid_nan_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_nan_atom(value);
+  if (!ok) { return errc; }
+  tape.append_double(std::numeric_limits<double>::quiet_NaN());
+  return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_nan_atom(json_iterator &iter, const uint8_t *value, error_code errc) noexcept {
+  iter.log_value("nan");
+  if (!atomparsing::is_valid_nan_atom(value, iter.remaining_len())) { return errc; }
+  tape.append_double(std::numeric_limits<double>::quiet_NaN());
+  return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_inf_atom(json_iterator &iter, const uint8_t *value) noexcept {
+  iter.log_value("inf");
+  // Because 'inf' is an extension, non a canonical atom, a tape error should be returned on failure.
+  // For unpadded input use the length-aware validator so the 'infinity' 8-byte
+  // compare cannot read past the buffer on a malformed token at the end.
+  const bool ok = UNPADDED ? atomparsing::is_valid_inf_atom(value, iter.remaining_len())
+                           : atomparsing::is_valid_inf_atom(value);
+  if (!ok) { return TAPE_ERROR; }
+  tape.append_double(std::numeric_limits<double>::infinity());
+  return SUCCESS;
+}
+
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::visit_root_inf_atom(json_iterator &iter, const uint8_t *value) noexcept {
+  iter.log_value("inf");
+  // Because 'inf' is an extension, non a canonical atom, a tape error should be returned on failure
+  if (!atomparsing::is_valid_inf_atom(value, iter.remaining_len())) { return TAPE_ERROR; }
+  tape.append_double(std::numeric_limits<double>::infinity());
+  return SUCCESS;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
 // private:

-simdjson_inline uint32_t tape_builder::next_tape_index(json_iterator &iter) const noexcept {
+template <bool UNPADDED>
+simdjson_inline uint32_t tape_builder_impl<UNPADDED>::next_tape_index(json_iterator &iter) const noexcept {
   return uint32_t(tape.next_tape_loc - iter.dom_parser.doc->tape.get());
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::empty_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
   auto start_index = next_tape_index(iter);
   tape.append(start_index+2, start);
   tape.append(start_index, end);
   return SUCCESS;
 }

-simdjson_inline void tape_builder::start_container(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_inline void tape_builder_impl<UNPADDED>::start_container(json_iterator &iter) noexcept {
   iter.dom_parser.open_containers[iter.depth].tape_index = next_tape_index(iter);
   iter.dom_parser.open_containers[iter.depth].count = 0;
   tape.skip(); // We don't actually *write* the start element until the end.
 }

-simdjson_warn_unused simdjson_inline error_code tape_builder::end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
+template <bool UNPADDED>
+simdjson_warn_unused simdjson_inline error_code tape_builder_impl<UNPADDED>::end_container(json_iterator &iter, internal::tape_type start, internal::tape_type end) noexcept {
   // Write the ending tape element, pointing at the start location
   const uint32_t start_tape_index = iter.dom_parser.open_containers[iter.depth].tape_index;
   tape.append(start_tape_index, end);
@@ -64992,13 +80499,15 @@ simdjson_warn_unused simdjson_inline error_code tape_builder::end_container(json
   return SUCCESS;
 }

-simdjson_inline uint8_t *tape_builder::on_start_string(json_iterator &iter) noexcept {
+template <bool UNPADDED>
+simdjson_inline uint8_t *tape_builder_impl<UNPADDED>::on_start_string(json_iterator &iter) noexcept {
   // we advance the point, accounting for the fact that we have a NULL termination
   tape.append(current_string_buf_loc - iter.dom_parser.doc->string_buf.get(), internal::tape_type::STRING);
   return current_string_buf_loc + sizeof(uint32_t);
 }

-simdjson_inline void tape_builder::on_end_string(uint8_t *dst) noexcept {
+template <bool UNPADDED>
+simdjson_inline void tape_builder_impl<UNPADDED>::on_end_string(uint8_t *dst) noexcept {
   uint32_t str_length = uint32_t(dst - (current_string_buf_loc + sizeof(uint32_t)));
   // TODO check for overflow in case someone has a crazy string (>=4GB?)
   // But only add the overflow check when the document itself exceeds 4GB
@@ -65229,14 +80738,14 @@ simdjson_warn_unused simdjson_inline error_code scan() {
     // Primitive or invalid character (invalid characters will be checked in stage 2)
     } else {
       // Anything else, add the structural and go until we find the next one.
-      // We also stop on '"' so that an unclosed string still reaches
-      // validate_string(); a quote swallowed by the run would hide it. A
-      // quote cannot occur inside a valid primitive. We deliberately do not
-      // stop on every ESC_ASCII character: that also covers a backslash and the
-      // control characters, and ending the run there makes the fallback
-      // disagree with the SIMD kernels.
+      // We also stop on RS (0x1E) so that RFC 7464 json_sequence inputs
+      // like `\x1e"a"\x1e"b"` produce a separate structural for each RS
+      // rather than being absorbed into a single primitive run, and on '"'
+      // so that an unclosed string still reaches validate_string(). Neither
+      // can occur inside a valid primitive.
       add_structural();
-      while (idx+1<len && !char_is_space_or_operator(buf[idx+1]) && buf[idx+1] != '"') {
+      while (idx+1<len && !char_is_space_or_operator(buf[idx+1]) &&
+             buf[idx+1] != 0x1e && buf[idx+1] != '"') {
         idx++;
       };
     }
@@ -65291,6 +80800,75 @@ simdjson_warn_unused simdjson_inline error_code scan() {
     // doing.
     parser.structural_indexes[parser.n_structural_indexes] = uint32_t(len);
     if (parser.n_structural_indexes == 0) { return EMPTY; }
+  } else if (partial == stage1_mode::json_sequence_partial) {
+    // RFC 7464: use RS positions for batch boundaries
+    // A discarded unclosed string also caps how far the filter may scan.
+    size_t scan_len = len;
+    if(unclosed_string) {
+      parser.n_structural_indexes--;
+      if (simdjson_unlikely(parser.n_structural_indexes == 0)) { return CAPACITY; }
+      scan_len = parser.structural_indexes[parser.n_structural_indexes];
+    }
+    uint32_t next_batch_start = uint32_t(len);
+    auto new_structural_indexes = find_next_document_index_json_sequence(parser, len, false, next_batch_start, scan_len);
+    if (new_structural_indexes == DOCUMENT_TOO_LARGE) {
+      return CAPACITY;
+    }
+    if (new_structural_indexes == 0) {
+      // An EMPTY batch must still advance next_batch_start, or document_stream
+      // re-parses the same bytes forever. CAPACITY when it cannot.
+      if (next_batch_start == 0) { return CAPACITY; }
+      parser.n_structural_indexes = 0;
+      parser.structural_indexes[0] = next_batch_start;
+      return EMPTY;
+    }
+    parser.n_structural_indexes = new_structural_indexes;
+    parser.structural_indexes[parser.n_structural_indexes] = next_batch_start;
+  } else if (partial == stage1_mode::json_sequence_final) {
+    // RFC 7464: final batch, last document extends to EOF
+    // As above: a discarded unclosed string caps how far the filter may scan.
+    size_t scan_len = len;
+    if(unclosed_string) {
+      parser.n_structural_indexes--;
+      scan_len = parser.structural_indexes[parser.n_structural_indexes];
+    }
+    uint32_t next_batch_start = uint32_t(len);
+    parser.n_structural_indexes = find_next_document_index_json_sequence(parser, len, true, next_batch_start, scan_len);
+    // The filter compacted structural_indexes in place and restored the EOF
+    // sentinel past the compacted end, so the copy below is either the start
+    // of a truncated document or len, as in streaming_final.
+    parser.structural_indexes[parser.n_structural_indexes + 1] = parser.structural_indexes[parser.n_structural_indexes];
+    parser.structural_indexes[parser.n_structural_indexes] = uint32_t(len);
+    if (simdjson_unlikely(parser.n_structural_indexes == 0)) { return EMPTY; }
+  } else if (partial == stage1_mode::comma_delimited_partial) {
+    // Comma-delimited: filter root-level commas, use comma positions for batch boundaries
+    if(unclosed_string) {
+      parser.n_structural_indexes--;
+      if (simdjson_unlikely(parser.n_structural_indexes == 0)) { return CAPACITY; }
+    }
+    uint32_t next_batch_start = uint32_t(len);
+    auto new_structural_indexes = filter_comma_delimited(parser, len, false, next_batch_start);
+    if (new_structural_indexes == DOCUMENT_TOO_LARGE) {
+      return CAPACITY;
+    }
+    if (new_structural_indexes == 0) {
+      // An EMPTY batch must still advance next_batch_start, or document_stream
+      // re-parses the same bytes forever. CAPACITY when it cannot.
+      if (next_batch_start == 0) { return CAPACITY; }
+      parser.n_structural_indexes = 0;
+      parser.structural_indexes[0] = next_batch_start;
+      return EMPTY;
+    }
+    parser.n_structural_indexes = new_structural_indexes;
+    parser.structural_indexes[parser.n_structural_indexes] = next_batch_start;
+  } else if (partial == stage1_mode::comma_delimited_final) {
+    // Comma-delimited: final batch, last document extends to EOF
+    if(unclosed_string) { parser.n_structural_indexes--; }
+    uint32_t next_batch_start = uint32_t(len);
+    parser.n_structural_indexes = filter_comma_delimited(parser, len, true, next_batch_start);
+    parser.structural_indexes[parser.n_structural_indexes + 1] = parser.structural_indexes[parser.n_structural_indexes];
+    parser.structural_indexes[parser.n_structural_indexes] = uint32_t(len);
+    if (simdjson_unlikely(parser.n_structural_indexes == 0)) { return EMPTY; }
   } else if(unclosed_string) { error = UNCLOSED_STRING; }
   return error;
 }
diff --git a/deps/simdjson/simdjson.h b/deps/simdjson/simdjson.h
index 43fe09631c0..58041ff68a7 100644
--- a/deps/simdjson/simdjson.h
+++ b/deps/simdjson/simdjson.h
@@ -1,4 +1,4 @@
-/* auto-generated on 2026-09-04 16:04:31 -0400. version 4.6.11 Do not edit! */
+/* auto-generated on 2026-10-07 22:43:29 -0400. version 5.0.3 Do not edit! */
 /* including simdjson.h:  */
 /* begin file simdjson.h */
 #ifndef SIMDJSON_H
@@ -61,7 +61,9 @@
 #endif

 // C++ 26
-#if !defined(SIMDJSON_CPLUSPLUS26) && (SIMDJSON_CPLUSPLUS >= 202402L) // update when the standard is finalized
+// While C++26 is a working draft, compilers report 202400L in C++26 mode
+// (both GCC 16 and Clang 21 do). Update when the standard is finalized.
+#if !defined(SIMDJSON_CPLUSPLUS26) && (SIMDJSON_CPLUSPLUS >= 202400L)
 #define SIMDJSON_CPLUSPLUS26 1
 #endif

@@ -118,14 +120,48 @@
 #endif
 #endif

-// The current specification is unclear on how we detect
-// static reflection, both __cpp_lib_reflection and
-// __cpp_impl_reflection are proposed in the draft specification.
-// For now, we disable static reflect by default. It must be
-// specified at compiler time.
+// Static reflection.
+//
+// The reflection-based APIs (simdjson::to, document::get<T>, the builder,
+// compile-time JSON, annotations) need considerably more than the reflection
+// operator. We turn them on only when the compiler advertises all of:
+//
+//   P2996 reflection (^^, splicers, <meta>)  __cpp_impl_reflection,
+//                                            __cpp_lib_reflection
+//   P1306 expansion statements (template for) __cpp_expansion_statements
+//   P3491 std::define_static_string / _array  __cpp_lib_define_static
+//
+// Two further features we rely on have, as of this writing, no feature-test
+// macro of their own, so they cannot be checked directly:
+//
+//   P3394 annotations ([[=x]], std::meta::annotations_of) -- used for
+//         the annotations of simdjson/annotations.h (rename, skip, ...).
+//   P3289 consteval blocks (consteval { ... }) -- used by compile_time_json.
+//
+// Every implementation that defines the four macros above also implements
+// those two, so requiring the four is sufficient in practice. If that ever
+// stops being true, define SIMDJSON_STATIC_REFLECTION=0 to opt out.
+//
+// SIMDJSON_STATIC_REFLECTION may always be defined by the user (or by the
+// build system) to 0 or 1 to override the detection.
+//
+// Note that C++26 mode alone is not enough: GCC 16 requires -freflection,
+// and only then does it define __cpp_impl_reflection.
 #ifndef SIMDJSON_STATIC_REFLECTION
-#define SIMDJSON_STATIC_REFLECTION 0 // disabled by default.
+#if defined(SIMDJSON_CPLUSPLUS26) &&                                           \
+    defined(__cpp_impl_reflection) && __cpp_impl_reflection >= 202506L &&      \
+    defined(__cpp_lib_reflection) && __cpp_lib_reflection >= 202506L &&        \
+    defined(__cpp_expansion_statements) &&                                     \
+        __cpp_expansion_statements >= 202506L &&                               \
+    defined(__cpp_lib_define_static) && __cpp_lib_define_static >= 202506L
+// __cpp_lib_reflection is the feature-test macro for <meta>, so there is no
+// need for a separate __has_include check (which would have to be guarded for
+// compilers that lack __has_include).
+#define SIMDJSON_STATIC_REFLECTION 1
+#else
+#define SIMDJSON_STATIC_REFLECTION 0
 #endif
+#endif // SIMDJSON_STATIC_REFLECTION

 #if defined(__apple_build_version__)
 #if __apple_build_version__ < 14000000
@@ -158,6 +194,47 @@
 #define SIMDJSON_SUPPORTS_DESERIALIZATION 0
 #endif

+// The C++20 char8_t type (and std::u8string/std::u8string_view) is available.
+// Because all strings that simdjson produces are valid UTF-8, we can offer
+// char8_t variants of our string accessors when this macro is set.
+#if !defined(SIMDJSON_SUPPORTS_CHAR8_T)
+#if defined(__cpp_char8_t) && __cpp_char8_t >= 201811L
+#define SIMDJSON_SUPPORTS_CHAR8_T 1
+#else
+#define SIMDJSON_SUPPORTS_CHAR8_T 0
+#endif
+#endif // !defined(SIMDJSON_SUPPORTS_CHAR8_T)
+
+// The C++23 fixed-width floating-point types std::float32_t and std::float64_t
+// (<stdfloat>) are available. They are optional even in C++23: a compiler that
+// provides them predefines __STDCPP_FLOAT32_T__ and __STDCPP_FLOAT64_T__.
+// When these macros are set, we offer get_float32() and get_float64().
+#if !defined(SIMDJSON_SUPPORTS_FLOAT32_T)
+#if defined(__STDCPP_FLOAT32_T__) && defined(__has_include)
+#if __has_include(<stdfloat>)
+#define SIMDJSON_SUPPORTS_FLOAT32_T 1
+#endif
+#endif
+#ifndef SIMDJSON_SUPPORTS_FLOAT32_T
+#define SIMDJSON_SUPPORTS_FLOAT32_T 0
+#endif
+#endif // !defined(SIMDJSON_SUPPORTS_FLOAT32_T)
+
+#if !defined(SIMDJSON_SUPPORTS_FLOAT64_T)
+#if defined(__STDCPP_FLOAT64_T__) && defined(__has_include)
+#if __has_include(<stdfloat>)
+#define SIMDJSON_SUPPORTS_FLOAT64_T 1
+#endif
+#endif
+#ifndef SIMDJSON_SUPPORTS_FLOAT64_T
+#define SIMDJSON_SUPPORTS_FLOAT64_T 0
+#endif
+#endif // !defined(SIMDJSON_SUPPORTS_FLOAT64_T)
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T || SIMDJSON_SUPPORTS_FLOAT64_T
+#include <stdfloat>
+#endif
+

 #if !defined(SIMDJSON_CONSTEVAL)
 #if defined(__cpp_consteval) && __cpp_consteval >= 201811L && defined(__cpp_lib_constexpr_string) && __cpp_lib_constexpr_string >= 201907L
@@ -166,6 +243,18 @@
 #define SIMDJSON_CONSTEVAL 0
 #endif // defined(__cpp_consteval) && __cpp_consteval >= 201811L && defined(__cpp_lib_constexpr_string) && __cpp_lib_constexpr_string >= 201907L
 #endif // !defined(SIMDJSON_CONSTEVAL)
+
+// SIMDJSON_CONSTEXPR_STRING is 'constexpr' when the standard library supports
+// constexpr std::string (e.g., libstdc++ 12 or better), and empty otherwise. It
+// lets functions that build a std::string be constant expressions when possible
+// while still compiling against older standard libraries.
+#if !defined(SIMDJSON_CONSTEXPR_STRING)
+#if defined(__cpp_lib_constexpr_string) && __cpp_lib_constexpr_string >= 201907L
+#define SIMDJSON_CONSTEXPR_STRING constexpr
+#else
+#define SIMDJSON_CONSTEXPR_STRING
+#endif // defined(__cpp_lib_constexpr_string) && __cpp_lib_constexpr_string >= 201907L
+#endif // !defined(SIMDJSON_CONSTEXPR_STRING)
 #endif // SIMDJSON_COMPILER_CHECK_H
 /* end file simdjson/compiler_check.h */
 /* including simdjson/portability.h: #include "simdjson/portability.h" */
@@ -457,16 +546,86 @@ using std::size_t;
 #endif
 #endif

+#ifndef SIMDJSON_HAS_UNISTD_H
+#if defined(__unix__) || defined(__APPLE__) || defined(__linux__)
+#define SIMDJSON_HAS_UNISTD_H 1
+#else
+#define SIMDJSON_HAS_UNISTD_H 0
+#endif
+#endif
+
+// padded_memory_map availability.
+//
+// On POSIX platforms the class is always available: the implementation uses
+// `mmap` (and a trailing anonymous page for padding) from <sys/mman.h>.
+//
+// On Windows the class is disabled by default and must be explicitly
+// opted into by defining `SIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS=1`. Enabling
+// it requires:
+//   1. `<windows.h>` has been included *before* `<simdjson.h>` (so that
+//      this header can see the Win32 types and the `_WINDOWS_` include
+//      guard),
+//   2. the compilation targets Windows 10, version 1803 or later
+//      (i.e. `NTDDI_VERSION >= NTDDI_WIN10_RS4`, `0x0A000005`). This is
+//      required because the implementation relies on the modern memory
+//      APIs introduced with that version (`CreateFileMapping2` /
+//      `MapViewOfFile3`),
+//   3. the link step pulls in an import library that exports those APIs,
+//      typically `onecore.lib` (or `mincore.lib`).
+//
+// The `SIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS` CMake option arranges (1)-(3)
+// automatically when building simdjson with its own CMake. Consumers using
+// simdjson as a pre-built library are responsible for setting the macro,
+// the Windows version macros, and the link library themselves.
+//
+// If the opt-in conditions are not met on Windows, `padded_memory_map`
+// simply does not exist -- any attempt to use it fails at compile time
+// with an "unknown identifier" diagnostic rather than silently degrading.
+//
+// The SIMDJSON_HAS_PADDED_MEMORY_MAP macro reflects whether the class is
+// available in the current translation unit. Users may test this macro to
+// conditionally compile code that depends on padded_memory_map.
+#ifndef SIMDJSON_HAS_PADDED_MEMORY_MAP
+  #if defined(__unix__) || defined(__APPLE__) || defined(__linux__)
+    #define SIMDJSON_HAS_PADDED_MEMORY_MAP 1
+  #elif defined(_WINDOWS_) && defined(SIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS) && SIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS
+    #define SIMDJSON_HAS_PADDED_MEMORY_MAP 1
+  #else
+    #define SIMDJSON_HAS_PADDED_MEMORY_MAP 0
+  #endif
+#endif

 #endif // SIMDJSON_PORTABILITY_H
 /* end file simdjson/portability.h */
+#include <cstddef>

 namespace simdjson {
 namespace internal {
+/**
+ * @private
+ * Scratch capacity that every caller of to_chars must provide.
+ *
+ * The emitted decimal is at most ~24 characters, but dragonbox() and
+ * format_buffer() intentionally write past the logical end with fixed-size
+ * 16/17-byte memcpy/memset operations so the compiler can inline them (no
+ * libc mem* dispatch with size-class branches). The extra bytes are required
+ * for safety of those over-writes; do not shrink this below 40.
+ * See src/to_chars.cpp and #2805.
+ */
+// Use an unscoped enum (not static constexpr / inline constexpr):
+// - C++11 targets (readme_examples11, quickstart11, ...) still include this header
+// - a static constexpr in the amalgamated simdjson.cpp TU is unused there
+//   (only callers in headers use it) and trips -Wunused-const-variable -Werror
+enum : size_t { to_chars_buffer_size = 40 };
 /**
  * @private
  * Our own implementation of the C++17 to_chars function.
  * Defined in src/to_chars
+ *
+ * @note The buffer starting at first must have at least to_chars_buffer_size
+ *       bytes of writable storage (see to_chars_buffer_size).
+ * @note The input number must be finite (NaN/Inf are not supported).
+ * @note The result is NOT null-terminated.
  */
 char *to_chars(char *first, const char *last, double value);
 /**
@@ -476,6 +635,12 @@ char *to_chars(char *first, const char *last, double value);
  */
 double from_chars(const char *first) noexcept;
 double from_chars(const char *first, const char* end) noexcept;
+/**
+ * @private
+ * Same as from_chars, but produces a correctly rounded binary32 (float) value.
+ * Defined in src/from_chars
+ */
+float from_chars_float(const char *first) noexcept;
 }

 #ifndef SIMDJSON_EXCEPTIONS
@@ -486,6 +651,10 @@ double from_chars(const char *first, const char* end) noexcept;
 #endif
 #endif

+#ifndef SIMDJSON_ENABLE_NAN_INF
+#define SIMDJSON_ENABLE_NAN_INF 0
+#endif
+
 } // namespace simdjson

 #if defined(__GNUC__)
@@ -501,16 +670,14 @@ double from_chars(const char *first, const char* end) noexcept;

 // Align to N-byte boundary
 #define SIMDJSON_ROUNDUP_N(a, n) (((a) + ((n)-1)) & ~((n)-1))
-#define SIMDJSON_ROUNDDOWN_N(a, n) ((a) & ~((n)-1))
-
-#define SIMDJSON_ISALIGNED_N(ptr, n) (((uintptr_t)(ptr) & ((n)-1)) == 0)

 #if SIMDJSON_REGULAR_VISUAL_STUDIO
   // We could use [[deprecated]] but it requires C++14
   #define simdjson_deprecated __declspec(deprecated)

   #define simdjson_really_inline __forceinline
-  #define simdjson_never_inline __declspec(noinline)
+  #define simdjson_never_inline inline __declspec(noinline)
+  #define simdjson_really_flatten [[msvc::flatten]]

   #define simdjson_unused
   #define simdjson_warn_unused
@@ -551,6 +718,7 @@ double from_chars(const char *first, const char* end) noexcept;

   #define simdjson_really_inline inline __attribute__((always_inline))
   #define simdjson_never_inline inline __attribute__((noinline))
+  #define simdjson_really_flatten [[gnu::flatten]]

   #define simdjson_unused __attribute__((unused))
   #define simdjson_warn_unused __attribute__((warn_unused_result))
@@ -627,6 +795,15 @@ double from_chars(const char *first, const char* end) noexcept;
   #define simdjson_inline simdjson_really_inline
 #endif

+#if defined(simdjson_flatten)
+  // Prefer the user's definition of simdjson_flatten; don't define it ourselves.
+#elif (defined(__GNUC__) && !defined(__OPTIMIZE__)) || (defined(_DEBUG) && _MSC_VER )
+  // Flattening can lead to significant code bloat and high compile times. Don't use it for unoptimized builds.
+  #define simdjson_flatten
+#else
+  #define simdjson_flatten simdjson_really_flatten
+#endif
+
 #if SIMDJSON_VISUAL_STUDIO
     /**
      * Windows users need to do some extra work when building
@@ -2538,22 +2715,22 @@ namespace std {
 #define SIMDJSON_SIMDJSON_VERSION_H

 /** The version of simdjson being used (major.minor.revision) */
-#define SIMDJSON_VERSION "4.6.11"
+#define SIMDJSON_VERSION "5.0.3"

 namespace simdjson {
 enum {
   /**
    * The major version (MAJOR.minor.revision) of simdjson being used.
    */
-  SIMDJSON_VERSION_MAJOR = 4,
+  SIMDJSON_VERSION_MAJOR = 5,
   /**
    * The minor version (major.MINOR.revision) of simdjson being used.
    */
-  SIMDJSON_VERSION_MINOR = 6,
+  SIMDJSON_VERSION_MINOR = 0,
   /**
    * The revision (major.minor.REVISION) of simdjson being used.
    */
-  SIMDJSON_VERSION_REVISION = 11
+  SIMDJSON_VERSION_REVISION = 3
 };
 } // namespace simdjson

@@ -2625,6 +2802,7 @@ enum error_code {
   OUT_OF_BOUNDS,              ///< Attempted to access location outside of document.
   TRAILING_CONTENT,           ///< Unexpected trailing content in the JSON input
   OUT_OF_CAPACITY,            ///< The capacity was exceeded, we cannot allocate enough memory.
+  UNKNOWN_FIELD,              ///< JSON field does not map to any member of the target (see simdjson::deny_unknown_fields)
   NUM_ERROR_CODES             ///< Placeholder for end of error code list.
 };

@@ -2987,6 +3165,7 @@ inline const std::string error_message(int error) noexcept;
 #if SIMDJSON_SUPPORTS_CONCEPTS

 #include <concepts>
+#include <string_view>
 #include <type_traits>

 namespace simdjson {
@@ -3028,6 +3207,19 @@ concept constructible_from_string_view = std::is_constructible_v<T, std::string_
                                         && !std::is_same_v<T, std::string_view>
                                         && std::is_default_constructible_v<T>;

+#if SIMDJSON_SUPPORTS_CHAR8_T
+/**
+ * A C++20 char8_t string type such as std::u8string. Such types cannot be built
+ * from a std::string_view (the character types differ), so they need their own
+ * deserialization path, going through the u8 string accessors.
+ */
+template<typename T>
+concept constructible_from_u8string_view = std::is_constructible_v<T, std::u8string_view>
+                                        && !std::is_same_v<T, std::u8string_view>
+                                        && !std::is_constructible_v<T, std::string_view>
+                                        && std::is_default_constructible_v<T>;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 template<typename M>
 concept string_view_keyed_map = string_view_like<typename M::key_type>
               && requires(std::remove_cvref_t<M>& m, typename M::key_type sv, typename M::mapped_type v) {
@@ -3122,9 +3314,15 @@ concept string_like =
 // Concept that checks if a type is a container but not a string (because
 // strings handling must be handled differently)
 // Now uses iterator-based approach for broader container support
+//
+// Optional types are excluded on purpose. Since C++26 (P3168), std::optional
+// is itself a range, so without the exclusion an std::optional would match
+// both this concept and optional_type, making the container and the optional
+// overloads of atom()/append() ambiguous. See issue 2827.
 template <typename T>
 concept container_but_not_string =
-  std::ranges::input_range<T> && !string_like<T> && !concepts::string_view_keyed_map<T>;
+  std::ranges::input_range<T> && !string_like<T> && !concepts::string_view_keyed_map<T>
+  && !concepts::optional_type<T>;



@@ -3231,6 +3429,11 @@ struct fixed_string {
             data[i] = str[i];
         }
     }
+    constexpr fixed_string(const unsigned char (&str)[N])  {
+        for (std::size_t i = 0; i < N; ++i) {
+            data[i] = static_cast<char>(str[i]);
+        }
+    }
     char data[N];
     constexpr std::string_view view() const { return {data, N - 1}; }
     constexpr size_t size() const { return N ; }
@@ -3262,6 +3465,11 @@ struct string_constant {
 #endif // SIMDJSON_CONSTEVALUTIL_H
 /* end file simdjson/constevalutil.h */

+#if SIMDJSON_SUPPORTS_CHAR8_T
+#include <string>
+#include <string_view>
+#endif
+
 /**
  * @brief The top level simdjson namespace, containing everything the library provides.
  */
@@ -3271,6 +3479,10 @@ SIMDJSON_PUSH_DISABLE_UNUSED_WARNINGS

 /** The maximum document size supported by simdjson. */
 constexpr size_t SIMDJSON_MAXSIZE_BYTES = 0xFFFFFFFF;
+/** The maximum depth of nested objects and arrays supported by simdjson.
+ A depth of SIMDJSON_MAXSIZE_BYTES/2 is not reasonable and would be
+ adversarial, but it serves as an upper bound for validation purposes. */
+constexpr size_t SIMDJSON_MAX_DEPTH = SIMDJSON_MAXSIZE_BYTES/2;

 /**
  * The amount of padding needed in a buffer to parse JSON.
@@ -3296,6 +3508,30 @@ struct padded_string;
 class padded_string_view;
 enum class stage1_mode;

+/**
+ * Stream format for parse_many/iterate_many.
+ */
+enum class stream_format {
+  whitespace_delimited, ///< Whitespace-delimited JSON documents (default, includes NDJSON/JSONL)
+  json_sequence,        ///< RFC 7464 JSON text sequences (RS-delimited)
+  comma_delimited,      ///< Comma-separated JSON documents (e.g., `{...},{...},{...}`)
+  comma_delimited_array,///< A single JSON array whose elements are iterated as
+                        ///< comma-separated documents (e.g., `[{...},{...},{...}]`).
+                        ///< The parser strips the outer `[` / `]` plus any
+                        ///< surrounding JSON whitespace (space, tab, LF, CR)
+                        ///< and then behaves like `comma_delimited` over the
+                        ///< remaining bytes.
+  newline_delimited     ///< NDJSON/JSON Lines where each document occupies exactly
+                        ///< one line: documents are separated by line feeds and no
+                        ///< document contains a raw line feed. Same inputs as
+                        ///< `whitespace_delimited`, but the stronger guarantee lets
+                        ///< the parser find the end of a document without walking
+                        ///< it. On ondemand `iterate_many`, an unread remainder may
+                        ///< be skipped by jumping to the next line feed without
+                        ///< structure-validating that remainder. Use
+                        ///< `whitespace_delimited` if unsure.
+};
+
 namespace internal {

 template<typename T>
@@ -3306,6 +3542,52 @@ class tape_ref;
 struct value128;
 enum class tape_type;

+#if SIMDJSON_SUPPORTS_CHAR8_T
+/**
+ * Reinterpret a UTF-8 string as a C++20 std::u8string_view. No byte is copied
+ * or modified: char8_t and char have the same size, representation and
+ * alignment. Every string that simdjson produces is valid UTF-8, so this is a
+ * lossless view over the very same memory.
+ * @private
+ */
+simdjson_inline std::u8string_view as_u8string_view(std::string_view v) noexcept {
+  return std::u8string_view(reinterpret_cast<const char8_t *>(v.data()), v.size());
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
+/**
+ * Assign a UTF-8 string to a string-like receiver. The general case simply
+ * assigns the std::string_view: it covers std::string and any user type that
+ * can be assigned from a std::string_view.
+ * @private
+ */
+template <typename string_type>
+simdjson_inline void assign_utf8(string_type &receiver, std::string_view content) noexcept {
+  receiver = content;
+}
+
+#if SIMDJSON_SUPPORTS_CHAR8_T
+/**
+ * Assign a UTF-8 string to a char8_t-based string (e.g., std::u8string). This
+ * overload is more specialized than the general one, so overload resolution
+ * prefers it whenever the receiver holds char8_t.
+ * @private
+ */
+template <typename traits_type, typename allocator_type>
+simdjson_inline void assign_utf8(std::basic_string<char8_t, traits_type, allocator_type> &receiver, std::string_view content) noexcept {
+  receiver.assign(reinterpret_cast<const char8_t *>(content.data()), content.size());
+}
+
+/**
+ * Assign a UTF-8 string to a char8_t-based string view (e.g., std::u8string_view).
+ * @private
+ */
+template <typename traits_type>
+simdjson_inline void assign_utf8(std::basic_string_view<char8_t, traits_type> &receiver, std::string_view content) noexcept {
+  receiver = std::basic_string_view<char8_t, traits_type>(reinterpret_cast<const char8_t *>(content.data()), content.size());
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 } // namespace internal
 } // namespace simdjson

@@ -3599,7 +3881,12 @@ class document;
 * 3) The stream_final mode allows us to truncate final
 * unterminated strings. It is useful in conjunction with streaming_partial.
 */
-enum class stage1_mode { regular, streaming_partial, streaming_final};
+enum class stage1_mode {
+  regular,
+  streaming_partial, streaming_final,
+  json_sequence_partial, json_sequence_final,
+  comma_delimited_partial, comma_delimited_final
+};

 /**
  * Returns true if mode == streaming_partial or mode == streaming_final
@@ -3611,7 +3898,6 @@ inline bool is_streaming(stage1_mode mode) {
   // return (mode == stage1_mode::streaming_partial || mode == stage1_mode::streaming_final);
 }

-
 namespace internal {


@@ -3796,6 +4082,16 @@ public:
   /** Whether to store big integers as strings instead of returning BIGINT_ERROR */
   bool _number_as_string{false};

+  /**
+   * Whether the input buffer passed to parse() is *not* padded to len +
+   * SIMDJSON_PADDING bytes. When true, stage 2 string parsing avoids reading
+   * past buf+len (it finishes the final, near-the-end bytes from a small padded
+   * scratch buffer). This is set only by the no-padding DOM parse entry points
+   * (dom::parser::parse_unpadded); the default padded fast path leaves it false
+   * and is unaffected.
+   */
+  bool _unpadded{false};
+
 protected:

   // Declaring these so that subclasses can use them to implement their constructors.
@@ -4350,11 +4646,26 @@ inline std::ostream& operator<<(std::ostream& out, const padded_string& s) { ret
 inline std::ostream& operator<<(std::ostream& out, simdjson_result<padded_string> &s) noexcept(false) { return out << s.value(); }
 #endif

-
-#ifndef _WIN32
+#if SIMDJSON_HAS_PADDED_MEMORY_MAP
 /**
  * A class representing a memory-mapped file with padding.
- * It is only available on non-Windows platforms, as Windows has different APIs for memory mapping.
+ *
+ * On POSIX systems (Linux, macOS, BSD, ...), this uses `mmap` to map the file
+ * contents directly into memory, which is efficient for large files (no copy).
+ *
+ * On Windows, this class is disabled by default and must be opted into at
+ * build time by defining `SIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS=1`. When
+ * enabled, `<windows.h>` must also be included before `<simdjson.h>` and
+ * the compilation must target Windows 10, version 1803 or later. The
+ * Windows implementation uses the modern memory APIs (`VirtualAlloc2`,
+ * `CreateFileMapping2`, `MapViewOfFile3`) with the placeholder virtual
+ * memory mechanism to always achieve true zero-copy mapping with
+ * contiguous zero-filled padding.
+ *
+ * Either way, the resulting `padded_string_view` carries at least
+ * `SIMDJSON_PADDING` bytes of accessible zero-filled padding after the file
+ * content, so it can be consumed directly by the simdjson parsers (including
+ * `parse_many` / `iterate_many`).
  */
 class padded_memory_map {
 public:
@@ -4362,9 +4673,11 @@ public:
    * Create a new padded memory map for the given file.
    * After creating the memory map, you can call view() to get a padded_string_view of the file content.
    * The memory map will be automatically released when the padded_memory_map instance is destroyed.
-   * Note that the file content is not copied, so this is efficient for large files. However,
-   * the file must remain unchanged while the memory map is in use. In case of error (e.g., file not found,
-   * permission denied, etc.), the memory map will be invalid and view() will return an empty view.
+   * On POSIX systems, the file content is not copied, so this is efficient for large files.
+   * On Windows, the file is mapped into memory via `MapViewOfFile3` (zero-copy).
+   * In all cases, the file must remain unchanged while the memory map is in use.
+   * In case of error (e.g., file not found, permission denied, etc.), the memory map will be
+   * invalid and view() will return an empty view.
    * You can check if the memory map is valid by calling is_valid() before using view().
    *
    * @param filename the path to the file to memory-map.
@@ -4401,8 +4714,14 @@ private:
   padded_memory_map &operator=(const padded_memory_map &) = delete;
   const char *data{nullptr};
   size_t size{0};
+#ifdef _WIN32
+  // When the file ends near an allocation-granularity boundary, we use the
+  // placeholder API to append zero-filled padding pages. This pointer tracks
+  // that region so the destructor can release it with VirtualFree.
+  void *padding_view_{nullptr};
+#endif
 };
-#endif // _WIN32
+#endif // SIMDJSON_HAS_PADDED_MEMORY_MAP



@@ -4476,6 +4795,9 @@ simdjson_warn_unused error_code minify(const char *buf, size_t len, char *dst, s
 #include <memory>
 #include <string>
 #include <ostream>
+#if SIMDJSON_CPLUSPLUS17
+#include <variant>
+#endif

 namespace simdjson {

@@ -4540,6 +4862,68 @@ public:

 }; // padded_string_view

+/**
+ * Get the system's memory page size. By default, we return
+ * 4096 bytes, which is the most common page size. On systems
+ * where the page size is not a multiple of 4096 bytes, and not
+ * a unix-like system, nor Windows, this function may return an
+ * incorrect value.
+ *
+ * @return The page size in bytes.
+ */
+inline uint32_t get_page_size() noexcept;
+
+#if SIMDJSON_CPLUSPLUS17
+/**
+ * A padded_input is a wrapper around either a padded_string_view or a padded_string.
+ * It will automatically pad a string_view if it does not have sufficient padding
+ * up to the end of the memory page. Note that a requirement for this method to
+ * make sense is to be on a system with a page size of at least 4096 (which is
+ * universal except on some embedded systems).
+ */
+struct padded_input {
+    /**
+     * Construct a padded_input from a string_view. If the string_view does not have sufficient padding,
+     * the data will be copied into a padded_string and the padded_string_view will point to the
+     * padded_string's data. Otherwise, the padded_string_view will point to the original string_view's data.
+     */
+     inline explicit padded_input(std::string_view sv);
+    /**
+     * Construct a padded_input from a C-style string (length specified). If the string does not have sufficient padding,
+     * the data will be copied into a padded_string and the padded_string_view will point to the
+     * padded_string's data. Otherwise, the padded_string_view will point to the original string's data.
+     */
+     inline explicit padded_input(const char *data, size_t length);
+    /**
+     * Construct a padded_input from a std::string. If the string does not have sufficient padding
+     * (considering its capacity), the data will be copied into a padded_string and the padded_string_view
+     * will point to the padded_string's data. Otherwise, the padded_string_view will point to the
+     * original string's data.
+     */
+     inline explicit padded_input(const std::string &s);
+
+    /**
+     * Check if the padded_input is a view.
+     *
+     * @return true if the padded_input is a view, false otherwise.
+     */
+     inline bool is_view() const noexcept;
+
+    /**
+     * Convert the padded_input to a padded_string_view.
+     *
+     * @return The padded_string_view.
+     */
+     inline operator simdjson::padded_string_view() const noexcept;
+
+private:
+    std::variant<simdjson::padded_string_view, simdjson::padded_string> storage;
+    // whether we cross a page boundary and need to allocate a new padded string.
+    static inline bool needs_allocation(const char* buf, size_t len, size_t padding = SIMDJSON_PADDING) noexcept;
+};
+
+#endif // SIMDJSON_CPLUSPLUS17
+
 #if SIMDJSON_EXCEPTIONS
 /**
  * Send padded_string instance to an output stream.
@@ -4572,6 +4956,31 @@ inline padded_string_view pad(std::string& s) noexcept;
  * @return The padded string.
  */
 inline padded_string_view pad_with_reserve(std::string& s) noexcept;
+
+/**
+ * Return the index-th document-aligned slice of a delimited stream.
+ *
+ * The input is divided into blocks of block_size bytes and each boundary is
+ * moved forward to just past the next delimiter, so a document is never split.
+ * The delimiter must not occur inside a document: a line feed for NDJSON, a
+ * record separator (0x1E) for RFC 7464.
+ *
+ * Slices are contiguous and non-overlapping, and each may be parsed
+ * independently, so callers can process them on as many threads as they like.
+ *
+ * Iterate while index * block_size < data.size(). A slice is empty when its
+ * block falls entirely inside one document, which happens only if that document
+ * is longer than block_size; skip it and continue. With block_size larger than
+ * the longest document, no slice is ever empty.
+ *
+ * @param data       The padded input.
+ * @param delimiter  The byte that separates documents.
+ * @param block_size The nominal slice size, before snapping.
+ * @param index      Which slice to return, counting from zero.
+ */
+inline padded_string_view slice_at(padded_string_view data, char delimiter,
+                                   size_t block_size, size_t index) noexcept;
+
 } // namespace simdjson

 #endif // SIMDJSON_PADDED_STRING_VIEW_H
@@ -4588,6 +4997,15 @@ inline padded_string_view pad_with_reserve(std::string& s) noexcept;

 #include <cstring> /* memcmp */

+// for page size computation.
+#if SIMDJSON_HAS_UNISTD_H
+  #include <unistd.h>
+  #if defined(__APPLE__)
+    #include <sys/sysctl.h>
+  #endif
+#endif
+
+
 namespace simdjson {

 inline padded_string_view::padded_string_view(const char* s, size_t len, size_t capacity) noexcept
@@ -4677,7 +5095,102 @@ inline padded_string_view pad_with_reserve(std::string& s) noexcept {
   return padded_string_view(s.data(), s.size(), s.capacity());
 }

+inline uint32_t get_page_size() noexcept {
+#if defined(_WINDOWS_) // if and only if someone loaded Windows.h, we can get the page size from there.
+// Otherwise, we assume 4096.
+    static const uint32_t cached = []() -> uint32_t {
+      SYSTEM_INFO si;
+      GetSystemInfo(&si);
+      return static_cast<std::uint32_t>(si.dwPageSize);
+    }();
+    return cached;
+#elif SIMDJSON_HAS_UNISTD_H
+    static const uint32_t cached = []() -> uint32_t {
+      long page_size = sysconf(_SC_PAGESIZE);
+      if (page_size > 0) {
+          return static_cast<uint32_t>(page_size);
+      }
+      return 4096; // fallback
+    }();
+    return cached;
+#else
+    return 4096; // fallback
+#endif
+}
+#if SIMDJSON_CPLUSPLUS17
+
+inline padded_input::padded_input(std::string_view sv)
+    : storage(simdjson::padded_string_view{}) {
+  if (needs_allocation(sv.data(), sv.size())) {
+      storage = simdjson::padded_string(sv);
+  } else {
+      storage = simdjson::padded_string_view(
+          sv.data(), sv.size(), sv.size() + simdjson::SIMDJSON_PADDING);
+  }
+}
+
+inline padded_input::padded_input(const char *data, size_t length)
+    : storage(simdjson::padded_string_view{}) {
+  if (needs_allocation(data, length)) {
+      storage = simdjson::padded_string(data, length);
+  } else {
+      storage = simdjson::padded_string_view(
+          data, length, length + simdjson::SIMDJSON_PADDING);
+  }
+}

+inline padded_input::padded_input(const std::string &s)
+    : storage(simdjson::padded_string_view{}) {
+  const size_t len = s.size();
+  const size_t cap = s.capacity();
+  // Here we have the string content from data() to data() + size(),
+  // but the memory is accessible from data() to data() + capacity().
+  const size_t needed_padding = (cap - len) < simdjson::SIMDJSON_PADDING
+    ? simdjson::SIMDJSON_PADDING - (cap - len) : 0;
+  if (needed_padding > 0 && needs_allocation(s.data(), cap, needed_padding)) {
+      storage = simdjson::padded_string(s);
+  } else {
+      storage = simdjson::padded_string_view(
+          s.data(), len, len + simdjson::SIMDJSON_PADDING);
+  }
+}
+
+inline bool padded_input::is_view() const noexcept {
+  return std::holds_alternative<simdjson::padded_string_view>(storage);
+}
+
+inline padded_input::operator simdjson::padded_string_view() const noexcept {
+  return std::visit([](const auto& p) -> simdjson::padded_string_view {
+      return p;
+  }, storage);
+}
+
+inline bool padded_input::needs_allocation(const char* buf, size_t len, size_t padding) noexcept {
+  if(len == 0) { return false; }
+  const auto page_size = get_page_size();
+  return ((reinterpret_cast<uintptr_t>(buf + len - 1) % page_size)
+          + padding >= static_cast<uintptr_t>(page_size));
+}
+#endif // SIMDJSON_CPLUSPLUS17
+
+inline padded_string_view slice_at(padded_string_view data, char delimiter,
+                                   size_t block_size, size_t index) noexcept {
+  if (block_size == 0 || index > data.size() / block_size) { return {}; }
+  const size_t raw_begin = index * block_size;
+  if (raw_begin >= data.size()) { return {}; }
+
+  auto snap = [&](size_t want) -> size_t {
+    if (want >= data.size()) { return data.size(); }
+    const void *p = std::memchr(data.data() + want, delimiter, data.size() - want);
+    return p ? size_t(static_cast<const char *>(p) - data.data()) + 1 : data.size();
+  };
+
+  const size_t begin = (raw_begin == 0) ? 0 : snap(raw_begin);
+  const size_t end = snap(raw_begin + block_size);
+  if (begin >= end) { return {}; }
+  return padded_string_view(data.data() + begin, end - begin,
+                            data.capacity() - begin);
+}

 } // namespace simdjson

@@ -4688,13 +5201,18 @@ inline padded_string_view pad_with_reserve(std::string& s) noexcept {
 #include <climits>
 #include <cwchar>

-#ifndef _WIN32
+#if SIMDJSON_HAS_UNISTD_H
 #include <fcntl.h>
 #include <stdio.h>
 #include <sys/mman.h>
 #include <sys/stat.h>
 #include <unistd.h>
 #endif
+// On Windows, `padded_memory_map` (when it is enabled) depends on types and
+// functions declared in <windows.h>. We deliberately do NOT include that
+// header here: users of simdjson who want `padded_memory_map` on Windows
+// must include <windows.h> themselves *before* including this header. See
+// padded_string.h for the detection logic.

 namespace simdjson {
 namespace internal {
@@ -5064,7 +5582,9 @@ inline bool padded_string_builder::reserve(size_t additional) noexcept {
 }


-#ifndef _WIN32
+#if SIMDJSON_HAS_PADDED_MEMORY_MAP
+
+#if SIMDJSON_HAS_UNISTD_H
 simdjson_inline padded_memory_map::padded_memory_map(const char *filename) noexcept {

     int fd = open(filename, O_RDONLY);
@@ -5100,7 +5620,132 @@ simdjson_inline padded_memory_map::~padded_memory_map() noexcept {
     munmap(const_cast<char *>(data), size + simdjson::SIMDJSON_PADDING);
   }
 }
+#elif defined(_WIN32)
+// Windows zero-copy implementation using placeholder virtual memory.
+//
+// We use the modern Windows memory APIs (VirtualAlloc2, CreateFileMapping2,
+// MapViewOfFile3 -- available since Windows 10 1803) to map the file into a
+// contiguous virtual address range that includes at least SIMDJSON_PADDING
+// zero bytes after the file content, with no data copies.
+//
+// Strategy:
+//   1. If rounding the file size up to the allocation granularity already
+//      exceeds file_size + SIMDJSON_PADDING, the OS page zero-fill provides
+//      the padding and we use a simple MapViewOfFile3 call.
+//   2. Otherwise we reserve a contiguous placeholder region via VirtualAlloc2,
+//      split it at the granularity-aligned file boundary, map the file into
+//      the first part, and commit zero pages for the second part (padding).
+simdjson_inline padded_memory_map::padded_memory_map(const char *filename) noexcept {
+  HANDLE file_handle = ::CreateFileA(
+      filename, GENERIC_READ,
+      FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE,
+      NULL, OPEN_EXISTING, FILE_ATTRIBUTE_NORMAL, NULL);
+  if (file_handle == INVALID_HANDLE_VALUE) {
+    return;
+  }
+  LARGE_INTEGER file_size_li;
+  if (!::GetFileSizeEx(file_handle, &file_size_li) || file_size_li.QuadPart < 0) {
+    ::CloseHandle(file_handle);
+    return;
+  }
+#if SIMDJSON_IS_32BITS
+  if (static_cast<unsigned long long>(file_size_li.QuadPart) >
+      static_cast<unsigned long long>(SIZE_MAX - simdjson::SIMDJSON_PADDING)) {
+    ::CloseHandle(file_handle);
+    return;
+  }
+#endif
+  size = static_cast<size_t>(file_size_li.QuadPart);
+  if (size == 0) {
+    ::CloseHandle(file_handle);
+    return;
+  }
+
+  HANDLE section = ::CreateFileMapping2(
+      file_handle, NULL, FILE_MAP_READ, PAGE_READONLY,
+      0, 0, NULL, NULL, 0);
+  ::CloseHandle(file_handle);
+  if (section == NULL) {
+    return;
+  }
+
+  SYSTEM_INFO si;
+  ::GetSystemInfo(&si);
+  const size_t granularity = static_cast<size_t>(si.dwAllocationGranularity);
+  const size_t file_region = (size + granularity - 1) & ~(granularity - 1);
+  const size_t total_needed = size + simdjson::SIMDJSON_PADDING;
+
+  if (file_region >= total_needed) {
+    // The zero-fill in the last page already covers the padding.
+    PVOID view = ::MapViewOfFile3(
+        section, ::GetCurrentProcess(), NULL, 0, 0,
+        0, PAGE_READONLY, NULL, 0);
+    ::CloseHandle(section);
+    if (view != NULL) {
+      data = static_cast<const char *>(view);
+    }
+    return;
+  }
+
+  // We need extra zero pages beyond the file region. Use the placeholder API
+  // to get a contiguous virtual address range spanning both the file mapping
+  // and the zero-filled padding.
+  const size_t padding_region =
+      ((total_needed - file_region) + granularity - 1) & ~(granularity - 1);
+  const size_t reserve_size = file_region + padding_region;
+
+  // Reserve a contiguous placeholder.
+  PVOID placeholder = ::VirtualAlloc2(
+      ::GetCurrentProcess(), NULL, reserve_size,
+      MEM_RESERVE | MEM_RESERVE_PLACEHOLDER, PAGE_NOACCESS, NULL, 0);
+  if (placeholder == NULL) {
+    ::CloseHandle(section);
+    return;
+  }

+  // Split into two placeholders at the file_region boundary.
+  if (!::VirtualFree(placeholder, file_region,
+                     MEM_RELEASE | MEM_PRESERVE_PLACEHOLDER)) {
+    ::VirtualFree(placeholder, 0, MEM_RELEASE);
+    ::CloseHandle(section);
+    return;
+  }
+
+  // Map the file into the first placeholder.
+  PVOID file_view = ::MapViewOfFile3(
+      section, ::GetCurrentProcess(), placeholder, 0, file_region,
+      MEM_REPLACE_PLACEHOLDER, PAGE_READONLY, NULL, 0);
+  ::CloseHandle(section);
+  if (file_view == NULL) {
+    ::VirtualFree(placeholder, 0, MEM_RELEASE);
+    ::VirtualFree(static_cast<char *>(placeholder) + file_region,
+                  0, MEM_RELEASE);
+    return;
+  }
+
+  // Commit zero pages in the second placeholder (the padding).
+  void *pad = static_cast<char *>(placeholder) + file_region;
+  PVOID padding_ptr = ::VirtualAlloc2(
+      ::GetCurrentProcess(), pad, padding_region,
+      MEM_REPLACE_PLACEHOLDER | MEM_COMMIT, PAGE_READONLY, NULL, 0);
+  if (padding_ptr == NULL) {
+    ::UnmapViewOfFile(file_view);
+    ::VirtualFree(pad, 0, MEM_RELEASE);
+    return;
+  }
+
+  data = static_cast<const char *>(file_view);
+  padding_view_ = padding_ptr;
+}
+
+simdjson_inline padded_memory_map::~padded_memory_map() noexcept {
+  if (data == nullptr) { return; }
+  ::UnmapViewOfFile(data);
+  if (padding_view_ != nullptr) {
+    ::VirtualFree(padding_view_, 0, MEM_RELEASE);
+  }
+}
+#endif // POSIX or _WIN32

 simdjson_inline simdjson::padded_string_view padded_memory_map::view() const noexcept simdjson_lifetime_bound {
   if(!is_valid()) {
@@ -5112,7 +5757,8 @@ simdjson_inline simdjson::padded_string_view padded_memory_map::view() const noe
 simdjson_inline bool padded_memory_map::is_valid() const noexcept {
   return data != nullptr;
 }
-#endif // _WIN32
+
+#endif // SIMDJSON_HAS_PADDED_MEMORY_MAP

 } // namespace simdjson

@@ -5222,6 +5868,9 @@ public:
   simdjson_inline tape_ref() noexcept;
   simdjson_inline tape_ref(const dom::document *doc, size_t json_index) noexcept;
   inline size_t after_element() const noexcept;
+  // The reference must point to an element boundary inside the array whose
+  // opening tag is at array_start, or to that array's closing tag.
+  inline size_t before_element(size_t array_start) const noexcept;
   simdjson_inline tape_type tape_ref_type() const noexcept;
   simdjson_inline uint64_t tape_value() const noexcept;
   simdjson_inline bool is_double() const noexcept;
@@ -5310,6 +5959,48 @@ public:
     friend class array;
   };

+  /**
+   * A forward iterator that visits the array's elements in reverse order.
+   * Like iterator, it returns element handles by value and does not own the
+   * document. No allocation or modification of the document is performed.
+   */
+  class reverse_iterator {
+  public:
+    using value_type = element;
+    using difference_type = std::ptrdiff_t;
+    using pointer = void;
+    using reference = value_type;
+    using iterator_category = std::forward_iterator_tag;
+
+    inline reference operator*() const noexcept;
+    inline reverse_iterator& operator++() noexcept;
+    inline reverse_iterator operator++(int) noexcept;
+    inline bool operator==(const reverse_iterator& other) const noexcept;
+    inline bool operator!=(const reverse_iterator& other) const noexcept;
+
+    reverse_iterator() noexcept = default;
+    reverse_iterator(const reverse_iterator&) noexcept = default;
+    reverse_iterator& operator=(const reverse_iterator&) noexcept = default;
+  private:
+    simdjson_inline reverse_iterator(const internal::tape_ref &tape, size_t array_start) noexcept;
+    internal::tape_ref tape{};
+    size_t array_start{};
+    friend class array;
+  };
+
+  /**
+   * Return the last array element, or rend() for an empty array.
+   * Incrementing the returned iterator moves toward the first element.
+   * A complete traversal takes O(n) time and O(1) additional space, where n
+   * is the number of immediate elements in the array.
+   * Finding the last or previous element can take O(n) time in the worst case when
+   * numeric payloads equal numeric type markers. Increments are amortized O(1)
+   * over a complete traversal; nested values do not increase this bound.
+   */
+  inline reverse_iterator rbegin() const noexcept;
+  /** Return the reverse traversal sentinel, before the first element. */
+  inline reverse_iterator rend() const noexcept;
+
   /**
    * Return the first array element.
    *
@@ -5445,6 +6136,8 @@ public:
 #if SIMDJSON_EXCEPTIONS
   inline dom::array::iterator begin() const noexcept(false);
   inline dom::array::iterator end() const noexcept(false);
+  inline dom::array::reverse_iterator rbegin() const noexcept(false);
+  inline dom::array::reverse_iterator rend() const noexcept(false);
   inline size_t size() const noexcept(false);
 #endif // SIMDJSON_EXCEPTIONS
 };
@@ -5820,6 +6513,64 @@ public:
   /** @private We do not want to allow implicit conversion from C string to std::string. */
   simdjson_inline simdjson_result<element> parse(const char *buf) noexcept = delete;

+  /**
+   * Parse a JSON document whose buffer is **not** padded, in place and without
+   * copying it.
+   *
+   * *This feature is currently experimental.*
+   *
+   * The standard parse() methods require the input buffer to have at least
+   * SIMDJSON_PADDING extra readable bytes after the document (or they copy it
+   * into a padded buffer when realloc_if_needed is true). parse_unpadded() lifts
+   * that requirement: it parses directly from your buffer of exactly `len` bytes,
+   * never reading past `buf + len`, and never allocating a full padded copy.
+   *
+   *   dom::parser parser;
+   *   std::string_view json = get_json(); // no trailing padding needed
+   *   dom::element doc = parser.parse_unpadded(json);
+   *
+   * This is the convenient way to use simdjson when you cannot (or do not want
+   * to) pad your input, e.g. a std::string_view into a larger buffer or a memory
+   * mapped file whose tail you do not control. It is generally a little slower
+   * than parsing a padded buffer with parse() (the very end of the document is
+   * handled with extra care), but it avoids the O(n) copy that
+   * parse(buf, len, true) performs when realloc_if_needed is true.
+   *
+   * The input is read but not modified, and it must remain valid (and the parser
+   * alive) for as long as you navigate the returned document, exactly like
+   * parse(buf, len, false).
+   *
+   * @param buf The JSON to parse. Only `len` bytes are read; no padding required.
+   * @param len The length of the JSON.
+   * @return An element pointing at the root of the document, or an error:
+   *         - MEMALLOC if the parser does not have enough capacity and allocation fails.
+   *         - CAPACITY if the parser does not have enough capacity and len > max_capacity.
+   *         - other json errors if parsing fails.
+   */
+  inline simdjson_result<element> parse_unpadded(const uint8_t *buf, size_t len) & noexcept;
+  inline simdjson_result<element> parse_unpadded(const uint8_t *buf, size_t len) && =delete;
+  /** @overload parse_unpadded(const uint8_t *buf, size_t len) */
+  simdjson_inline simdjson_result<element> parse_unpadded(const char *buf, size_t len) & noexcept;
+  simdjson_inline simdjson_result<element> parse_unpadded(const char *buf, size_t len) && =delete;
+  /** @overload parse_unpadded(const uint8_t *buf, size_t len) */
+  simdjson_inline simdjson_result<element> parse_unpadded(std::string_view s) & noexcept;
+  simdjson_inline simdjson_result<element> parse_unpadded(std::string_view s) && =delete;
+
+  /**
+   * Parse a non-padded JSON document into a caller-provided document instance, in
+   * place and without copying. This is to parse_unpadded() what
+   * parse_into_document() is to parse(). See parse_unpadded() for the padding and
+   * lifetime semantics.
+   *
+   * *This feature is currently experimental.*
+   *
+   * @param doc The document instance where the parsed data will be stored (on success).
+   * @param buf The JSON to parse. Only `len` bytes are read; no padding required.
+   * @param len The length of the JSON.
+   */
+  inline simdjson_result<element> parse_into_document_unpadded(document& doc, const uint8_t *buf, size_t len) & noexcept;
+  inline simdjson_result<element> parse_into_document_unpadded(document& doc, const uint8_t *buf, size_t len) && =delete;
+
   /**
    * Parse a JSON document into a provide document instance and return a temporary reference to it.
    * It is similar to the function `parse` except that instead of parsing into the internal
@@ -6045,7 +6796,7 @@ public:
    * @param batch_size The batch size to use. MUST be larger than the largest document. The sweet
    *                   spot is cache-related: small enough to fit in cache, yet big enough to
    *                   parse as many documents as possible in one tight loop.
-   *                   Defaults to 10MB, which has been a reasonable sweet spot in our tests.
+   *                   Defaults to 1MB, which has been a reasonable sweet spot in our tests.
    * @return The stream, or an error. An empty input will yield 0 documents rather than an EMPTY error. Errors:
    *         - MEMALLOC if the parser does not have enough capacity and memory allocation fails
    *         - CAPACITY if the parser does not have enough capacity and batch_size > max_capacity.
@@ -6057,14 +6808,50 @@ public:
   inline simdjson_result<document_stream> parse_many(const char *buf, size_t len, size_t batch_size = dom::DEFAULT_BATCH_SIZE) noexcept;
   /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
   inline simdjson_result<document_stream> parse_many(const std::string &s, size_t batch_size = dom::DEFAULT_BATCH_SIZE) noexcept;
-  inline simdjson_result<document_stream> parse_many(const std::string &&s, size_t batch_size) = delete;// unsafe
+  inline simdjson_result<document_stream> parse_many(const std::string &&s, size_t batch_size = dom::DEFAULT_BATCH_SIZE) = delete;// unsafe
   /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
   inline simdjson_result<document_stream> parse_many(const padded_string &s, size_t batch_size = dom::DEFAULT_BATCH_SIZE) noexcept;
-  inline simdjson_result<document_stream> parse_many(const padded_string &&s, size_t batch_size) = delete;// unsafe
+  inline simdjson_result<document_stream> parse_many(const padded_string &&s, size_t batch_size = dom::DEFAULT_BATCH_SIZE) = delete;// unsafe
+  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size)
+   *
+   * Because padded_string_view guarantees SIMDJSON_PADDING trailing bytes, this
+   * overload is safe to use with buffers that the caller owns elsewhere (for
+   * example, a padded_memory_map), with no extra copy. Without this overload,
+   * passing a padded_string_view would silently bind to the padded_string
+   * overload via an implicit conversion, allocating and copying the input, and
+   * -- because that temporary is destroyed at the end of the full-expression --
+   * leaving the returned document_stream pointing at freed memory. */
+  inline simdjson_result<document_stream> parse_many(const padded_string_view &v, size_t batch_size = dom::DEFAULT_BATCH_SIZE) noexcept;

   /** @private We do not want to allow implicit conversion from C string to std::string. */
   simdjson_result<document_stream> parse_many(const char *buf, size_t batch_size = dom::DEFAULT_BATCH_SIZE) noexcept = delete;

+  /**
+   * Parse a stream of JSON documents with explicit format specification.
+   *
+   * @param buf The concatenated JSON documents.
+   * @param len The length of the buffer.
+   * @param batch_size The batch size to use.
+   * @param format The stream format.
+   * @return A stream of documents, or an error.
+   */
+  inline simdjson_result<document_stream> parse_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> parse_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> parse_many(const std::string &s, size_t batch_size, stream_format format) noexcept;
+  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> parse_many(const padded_string &s, size_t batch_size, stream_format format) noexcept;
+  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> parse_many(const padded_string_view &v, size_t batch_size, stream_format format) noexcept;
+  /** @private An rvalue input is destroyed at the end of the full-expression, while
+   * the returned document_stream only holds a pointer to it: iterating the stream would
+   * then read freed memory. These deleted overloads also catch a std::string_view
+   * argument, which would otherwise convert implicitly to a padded_string temporary. */
+  inline simdjson_result<document_stream> parse_many(const std::string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+  /** @private @overload parse_many(const std::string &&s, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> parse_many(const padded_string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+
   /**
    * Ensure this parser has enough memory to process JSON documents up to `capacity` bytes in length
    * and `max_depth` depth.
@@ -6354,8 +7141,9 @@ public:
    *
    * IMPORTANT: this value is only meaningful under the conditions below. It is
    * computed from stage-1 bookkeeping, and outside these conditions it is not
-   * merely imprecise, it is arbitrary -- it can exceed size_in_bytes() or wrap
-   * around to a huge value. Check it only when both of the following hold:
+   * merely imprecise, it is
+   * arbitrary -- it can exceed size_in_bytes() or wrap around to a huge value.
+   * Check it only when all of the following hold:
    *
    *   - you iterated all the way to the end of the stream;
    *   - no document reported an error. Iteration stops at the first failed
@@ -6364,6 +7152,9 @@ public:
    * If you need to know about a truncated tail outside those conditions, track
    * it yourself from the last successful document (see iterator::current_index()
    * and iterator::source()).
+   *
+   * An empty input (zero bytes) or an input made only of white space contains
+   * no document: truncated_bytes() returns zero.
    */
   inline size_t truncated_bytes() const noexcept;
   /**
@@ -6463,12 +7254,14 @@ private:
    * @param buf is the raw byte buffer we need to process
    * @param len is the length of the raw byte buffer in bytes
    * @param batch_size is the size of the windows (must be strictly greater or equal to the largest JSON document)
+   * @param format is the stream format
    */
   simdjson_inline document_stream(
     dom::parser &parser,
     const uint8_t *buf,
     size_t len,
-    size_t batch_size
+    size_t batch_size,
+    stream_format format = stream_format::whitespace_delimited
   ) noexcept;

   /**
@@ -6518,6 +7311,8 @@ private:
   const uint8_t *buf;
   size_t len;
   size_t batch_size;
+  /** The stream format. */
+  stream_format format;
   /** The error (or lack thereof) from the current document. */
   error_code error;
   size_t batch_start{0};
@@ -6605,6 +7400,8 @@ enum class element_type {
   STRING = '"',    ///< std::string_view
   BOOL = 't',      ///< bool
   NULL_VALUE = 'n', ///< null
+  /// The BIGINT type is for integers that do not fit in 64 bits. It is only present
+  // if you set parser.number_as_string(true).
   BIGINT = 'Z'     ///< std::string_view: big integer stored as raw digit string
 };

@@ -6673,6 +7470,20 @@ public:
    *          Returns INCORRECT_TYPE if the JSON element is not a string.
    */
   inline simdjson_result<std::string_view> get_string() const noexcept;
+
+  #if SIMDJSON_SUPPORTS_CHAR8_T
+  /**
+   * Cast this element to a C++20 UTF-8 string.
+   *
+   * The string is guaranteed to be valid UTF-8.
+   *
+   * @returns A std::u8string_view. The string is stored in the parser and will be invalidated the next time it
+   *          parses a document or when it is destroyed.
+   *          Returns INCORRECT_TYPE if the JSON element is not a string.
+   */
+  inline simdjson_result<std::u8string_view> get_u8string() const noexcept;
+  #endif
+
   /**
    * Cast this element to a signed integer.
    *
@@ -6778,7 +7589,7 @@ public:
    * Supported types:
    * - Boolean: bool
    * - Number: double, uint64_t, int64_t
-   * - String: std::string_view, const char *
+   * - String: std::string_view, const char *, std::u8string_view (C++20)
    * - Array: dom::array
    * - Object: dom::object
    *
@@ -6793,7 +7604,7 @@ public:
    * Supported types:
    * - Boolean: bool
    * - Number: double, uint64_t, int64_t
-   * - String: std::string_view, const char *
+   * - String: std::string_view, const char *, std::u8string_view (C++20)
    * - Array: dom::array
    * - Object: dom::object
    *
@@ -6823,7 +7634,7 @@ public:
    * Supported types:
    * - Boolean: bool
    * - Number: double, uint64_t, int64_t
-   * - String: std::string_view, const char *
+   * - String: std::string_view, const char *, std::u8string_view (C++20)
    * - Array: dom::array
    * - Object: dom::object
    *
@@ -6842,7 +7653,7 @@ public:
    * Supported types:
    * - Boolean: bool
    * - Number: double, uint64_t, int64_t
-   * - String: std::string_view, const char *
+   * - String: std::string_view, const char *, std::u8string_view (C++20)
    * - Array: dom::array
    * - Object: dom::object
    *
@@ -7126,6 +7937,9 @@ public:
   simdjson_inline simdjson_result<const char *> get_c_str() const noexcept;
   simdjson_inline simdjson_result<size_t> get_string_length() const noexcept;
   simdjson_inline simdjson_result<std::string_view> get_string() const noexcept;
+  #if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string() const noexcept;
+  #endif
   simdjson_inline simdjson_result<int64_t> get_int64() const noexcept;
   simdjson_inline simdjson_result<uint64_t> get_uint64() const noexcept;
   simdjson_inline simdjson_result<double> get_double() const noexcept;
@@ -7825,6 +8639,24 @@ template <class T> std::string prettify(simdjson_result<T> x) {

 namespace simdjson {

+/** Specifies where commas should be in table-formatted elements. */
+enum class table_comma_placement {
+  /** Commas come right after the value */
+  before_padding,
+  /** Commas come after the column padding, so they line up in their own column. */
+  after_padding,
+  /** Commas come right after the value, except for columns of numbers */
+  before_padding_except_numbers,
+};
+
+/** Options for how lists or columns of numbers should be aligned */
+enum class number_list_alignment {
+  /** Left-aligns numbers */
+  left,
+  /** Right-aligns numbers */
+  right,
+};
+
 /**
  * Configuration options for FracturedJson formatting.
  *
@@ -7839,12 +8671,6 @@ struct fractured_json_options {
    */
   size_t max_total_line_length = 120;

-  /**
-   * Maximum length for inlined elements (default: 80).
-   * Simple arrays/objects shorter than this may be rendered inline.
-   */
-  size_t max_inline_length = 80;
-
   /**
    * Maximum nesting depth for inline rendering (default: 2).
    * Elements with complexity exceeding this will be expanded.
@@ -7853,11 +8679,11 @@ struct fractured_json_options {
   size_t max_inline_complexity = 2;

   /**
-   * Maximum complexity for compact array formatting (default: 1).
+   * Maximum complexity for compact array formatting (default: 2).
    * Arrays with elements of this complexity or less may have multiple
    * items per line.
    */
-  size_t max_compact_array_complexity = 1;
+  size_t max_compact_array_complexity = 2;

   /**
    * Number of spaces per indentation level (default: 4).
@@ -7865,23 +8691,26 @@ struct fractured_json_options {
   size_t indent_spaces = 4;

   /**
-   * Enable tabular formatting for arrays of similar objects (default: true).
-   * When enabled, arrays of objects with identical keys are formatted
-   * as aligned tables.
+   * Forces elements close to the root to always fully expand, regardless of other settings.
+   * (default: -1). -1 = none; 0 = root node only; 1 = root node and its children; etc.
    */
-  bool enable_table_format = true;
+  int always_expand_depth = -1;

   /**
-   * Minimum number of rows to trigger table mode (default: 3).
+   * Enable tabular formatting for arrays of similar objects or arrays
+   * (default: true). When enabled, the rows of such an array are written one
+   * per line with their columns aligned. Rows need not have identical keys:
+   * columns are ordered by the first occurrence of each key, and a row
+   * missing a key gets blank space in that column.
    */
-  size_t min_table_rows = 3;
+  bool enable_table_format = true;

   /**
-   * Similarity threshold for table detection (default: 0.8).
-   * Objects must share at least this fraction of keys to be formatted
-   * as a table.
+   * Maximum complexity of each row of a table (default: 2).
+   * 0 = rows may only be scalars (a single column); 1 = rows may be flat
+   * arrays/objects; higher values allow deeper nesting.
    */
-  double table_similarity_threshold = 0.8;
+  size_t max_table_row_complexity = 2;

   /**
    * Enable compact multiline arrays (default: true).
@@ -7891,16 +8720,26 @@ struct fractured_json_options {
   bool enable_compact_multiline = true;

   /**
-   * Maximum array items per line in compact mode (default: 10).
+   * Minimum number of items per line for an array to be formatted as a
+   * compact multiline array (default: 3).
    */
-  size_t max_items_per_line = 10;
+  size_t min_compact_array_row_items = 3;

   /**
-   * Add space inside brackets for simple containers (default: true).
-   * When true: { "key": "value" }
-   * When false: {"key": "value"}
+   * Add space inside brackets for containers that hold only scalar values
+   * (default: false). When true: { "key": "value" }. When false:
+   * {"key": "value"}.
+   * @see nested_bracket_padding
    */
-  bool simple_bracket_padding = true;
+  bool simple_bracket_padding = false;
+
+  /**
+   * Add space inside brackets for containers that hold at least one
+   * nested array/object (default: true). When true: { "a": [1, 2] }.
+   * When false: {"a": [1, 2]}.
+   * @see simple_bracket_padding
+   */
+  bool nested_bracket_padding = true;

   /**
    * Add space after colons (default: true).
@@ -7915,6 +8754,18 @@ struct fractured_json_options {
    * When false: [1,2,3]
    */
   bool comma_padding = true;
+
+  /**
+   * Placement of commas relative to column padding in table-formatted rows
+   * and compact multiline arrays (default: before_padding_except_numbers).
+   */
+  table_comma_placement comma_placement = table_comma_placement::before_padding_except_numbers;
+
+  /**
+   * Controls alignment of numbers in table columns or compact multiline arrays
+   * (default: left). Numbers are always written exactly as in the input.
+   */
+  number_list_alignment number_alignment = number_list_alignment::left;
 };

 /**
@@ -7995,12 +8846,55 @@ inline std::string fractured_json_string(std::string_view json_str,
 #ifndef SIMDJSON_JSONPATHUTIL_H
 #define SIMDJSON_JSONPATHUTIL_H

+/* skipped duplicate #include "simdjson/error.h" */
 #include <string>
 /* skipped duplicate #include "simdjson/common_defs.h" */

+#include <limits>
 #include <utility>

 namespace simdjson {
+namespace internal {
+/**
+ * Parses the next JSON Pointer array index token.
+ *
+ * The caller passes a pointer fragment with no leading '/', such as "123/foo".
+ * On success, array_index receives the parsed index and token_length receives
+ * the number of bytes consumed before the next '/' or the end of the fragment.
+ */
+simdjson_inline error_code parse_json_pointer_array_index(std::string_view json_pointer,
+                                                          size_t &array_index,
+                                                          size_t &token_length) noexcept {
+  array_index = 0;
+  token_length = 0;
+
+  for (; token_length < json_pointer.length() && json_pointer[token_length] != '/';
+       token_length++) {
+    uint8_t digit = uint8_t(json_pointer[token_length] - '0');
+    // Check for non-digit in array index. If it's there, we're trying to get a field in an object.
+    if (digit > 9) {
+      return INCORRECT_TYPE;
+    }
+    // 0 followed by other digits is invalid.
+    if (token_length > 0 && json_pointer[0] == '0') {
+      return INVALID_JSON_POINTER;
+    }
+    if (array_index >
+        (((std::numeric_limits<size_t>::max)() - digit) / 10)) {
+      return INDEX_OUT_OF_BOUNDS;
+    }
+    array_index = array_index * 10 + digit;
+  }
+
+  // Empty string is invalid; so is a "/" with no digits before it.
+  if (token_length == 0) {
+    return INVALID_JSON_POINTER;
+  }
+
+  return SUCCESS;
+}
+} // namespace internal
+
 /**
  * Converts JSONPath to JSON Pointer.
  * @param json_path The JSONPath string to be converted.
@@ -8220,6 +9114,72 @@ inline size_t tape_ref::after_element() const noexcept {
 simdjson_inline tape_type tape_ref::tape_ref_type() const noexcept {
   return static_cast<tape_type>(doc->tape[json_index] >> 56);
 }
+simdjson_inline size_t tape_ref::before_element(size_t array_start) const noexcept {
+  SIMDJSON_DEVELOPMENT_ASSERT(usable());
+  SIMDJSON_DEVELOPMENT_ASSERT(json_index > array_start);
+  tape_ref previous(doc, json_index - 1);
+  if (previous.json_index == array_start) { return array_start; }
+  // An exact numeric marker cannot end an element unless it is itself the
+  // payload of a number. In that case its header is immediately before it.
+  if (previous.is_int64() || previous.is_uint64() || previous.is_double()) {
+    return previous.json_index - 1;
+  }
+  tape_ref probe(doc, previous.json_index - 1);
+
+  // Validate both container links before examining its contents. A candidate
+  // opening tag preceded by an even run of numeric markers is a real tag,
+  // not a numeric payload. Scan both candidate boundaries together so that a
+  // forged opening tag inside a nested value cannot cause an unbounded detour.
+  // For a real container, the opening probe visits preceding siblings. For a
+  // numeric payload, the other probe does. Stopping at the shorter run bounds
+  // the work by the array's immediate elements, rather than nested contents.
+  const auto type = previous.tape_ref_type();
+  if (type == tape_type::END_ARRAY || type == tape_type::END_OBJECT) {
+    const size_t start = previous.matching_brace_index();
+    if (start > array_start && start < previous.json_index) {
+      tape_ref opening(doc, start);
+      const auto expected = type == tape_type::END_ARRAY
+          ? tape_type::START_ARRAY : tape_type::START_OBJECT;
+      if (opening.tape_ref_type() == expected &&
+          opening.matching_brace_index() == previous.json_index + 1) {
+        tape_ref before_opening(doc, start - 1);
+        while ((before_opening.is_int64() || before_opening.is_uint64() ||
+                before_opening.is_double()) &&
+               (probe.is_int64() || probe.is_uint64() || probe.is_double())) {
+          --before_opening.json_index;
+          --probe.json_index;
+        }
+        if (!before_opening.is_int64() && !before_opening.is_uint64() &&
+            !before_opening.is_double() &&
+            (start - before_opening.json_index) % 2 == 1) {
+          return start;
+        }
+      }
+    }
+  }
+
+  // Numeric payloads can have ANY bit pattern, including another type's tag.
+  // A run of exact numeric markers starts with a header, then alternates
+  // between payload and header. An odd run before this word makes it a payload.
+  // Subsequent reverse increments through exact-marker payloads take the
+  // constant-time numeric-payload branch above.
+  while (probe.is_int64() || probe.is_uint64() || probe.is_double()) {
+    --probe.json_index;
+  }
+  if ((previous.json_index - probe.json_index) % 2 == 0) {
+    return previous.json_index - 1;
+  }
+
+  // Once distinguished from numeric payloads, closing container tags link
+  // directly back to their opening tags.
+  switch (previous.tape_ref_type()) {
+    case tape_type::END_ARRAY:
+    case tape_type::END_OBJECT:
+      return previous.matching_brace_index();
+    default:
+      return previous.json_index;
+  }
+}
 simdjson_inline uint64_t internal::tape_ref::tape_value() const noexcept {
   return doc->tape[json_index] & internal::JSON_VALUE_MASK;
 }
@@ -8295,6 +9255,14 @@ inline size_t simdjson_result<dom::array>::size() const noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
   return first.size();
 }
+inline dom::array::reverse_iterator simdjson_result<dom::array>::rbegin() const noexcept(false) {
+  if (error()) { throw simdjson_error(error()); }
+  return first.rbegin();
+}
+inline dom::array::reverse_iterator simdjson_result<dom::array>::rend() const noexcept(false) {
+  if (error()) { throw simdjson_error(error()); }
+  return first.rend();
+}

 #endif // SIMDJSON_EXCEPTIONS

@@ -8304,6 +9272,7 @@ inline simdjson_result<dom::element> simdjson_result<dom::array>::at_pointer(std
 }

  inline simdjson_result<dom::element> simdjson_result<dom::array>::at_path(std::string_view json_path) const noexcept {
+  if (error()) { return error(); }
   auto json_pointer = json_path_to_pointer_conversion(json_path);
   if (json_pointer == "-1") { return INVALID_JSON_POINTER; }
   return at_pointer(json_pointer);
@@ -8344,6 +9313,15 @@ inline size_t array::size() const noexcept {
   SIMDJSON_DEVELOPMENT_ASSERT(tape.usable()); // https://github.com/simdjson/simdjson/issues/1914
   return tape.scope_count();
 }
+inline array::reverse_iterator array::rbegin() const noexcept {
+  SIMDJSON_DEVELOPMENT_ASSERT(tape.usable());
+  const internal::tape_ref end_tape(tape.doc, tape.matching_brace_index() - 1);
+  return reverse_iterator(internal::tape_ref(tape.doc, end_tape.before_element(tape.json_index)), tape.json_index);
+}
+inline array::reverse_iterator array::rend() const noexcept {
+  SIMDJSON_DEVELOPMENT_ASSERT(tape.usable());
+  return reverse_iterator(tape, tape.json_index);
+}
 inline size_t array::number_of_slots() const noexcept {
   SIMDJSON_DEVELOPMENT_ASSERT(tape.usable()); // https://github.com/simdjson/simdjson/issues/1914
   return tape.matching_brace_index() - tape.json_index;
@@ -8360,21 +9338,9 @@ inline simdjson_result<element> array::at_pointer(std::string_view json_pointer)
   // We don't support this, because we're returning a real element, not a position.
   if (json_pointer == "-") { return INDEX_OUT_OF_BOUNDS; }

-  // Read the array index
   size_t array_index = 0;
   size_t i;
-  for (i = 0; i < json_pointer.length() && json_pointer[i] != '/'; i++) {
-    uint8_t digit = uint8_t(json_pointer[i] - '0');
-    // Check for non-digit in array index. If it's there, we're trying to get a field in an object
-    if (digit > 9) { return INCORRECT_TYPE; }
-    array_index = array_index*10 + digit;
-  }
-
-  // 0 followed by other digits is invalid
-  if (i > 1 && json_pointer[0] == '0') { return INVALID_JSON_POINTER; } // "JSON pointer array index has other characters after 0"
-
-  // Empty string is invalid; so is a "/" with no digits before it
-  if (i == 0) { return INVALID_JSON_POINTER; } // "Empty string in JSON pointer array index"
+  SIMDJSON_TRY(internal::parse_json_pointer_array_index(json_pointer, array_index, i));

   // Get the child
   auto child = array(tape).at(array_index);
@@ -8409,7 +9375,6 @@ inline void array::process_json_path_of_child_elements(std::vector<element>::ite
     if(error) {
       continue;
     }
-    accumulator.reserve(accumulator.size() + child_result.size());
     accumulator.insert(accumulator.end(),
                         std::make_move_iterator(child_result.begin()),
                         std::make_move_iterator(child_result.end()));
@@ -8505,6 +9470,30 @@ inline array::operator element() const noexcept {
 	return element(tape);
 }

+//
+// array::reverse_iterator inline implementation
+//
+simdjson_inline array::reverse_iterator::reverse_iterator(const internal::tape_ref &_tape, size_t _array_start) noexcept
+    : tape{_tape}, array_start{_array_start} { }
+inline element array::reverse_iterator::operator*() const noexcept {
+  return element(tape);
+}
+inline array::reverse_iterator& array::reverse_iterator::operator++() noexcept {
+  tape.json_index = tape.before_element(array_start);
+  return *this;
+}
+inline array::reverse_iterator array::reverse_iterator::operator++(int) noexcept {
+  reverse_iterator out = *this;
+  ++*this;
+  return out;
+}
+inline bool array::reverse_iterator::operator==(const reverse_iterator& other) const noexcept {
+  return tape.doc == other.tape.doc && tape.json_index == other.tape.json_index;
+}
+inline bool array::reverse_iterator::operator!=(const reverse_iterator& other) const noexcept {
+  return !(*this == other);
+}
+
 //
 // array::iterator inline implementation
 //
@@ -8596,6 +9585,7 @@ inline simdjson_result<dom::element> simdjson_result<dom::object>::at_pointer(st
   return first.at_pointer(json_pointer);
 }
 inline simdjson_result<dom::element> simdjson_result<dom::object>::at_path(std::string_view json_path) const noexcept {
+  if (error()) { return error(); }
   auto json_pointer = json_path_to_pointer_conversion(json_path);
   if (json_pointer == "-1") { return INVALID_JSON_POINTER; }
   return at_pointer(json_pointer);
@@ -8725,7 +9715,6 @@ inline void object::process_json_path_of_child_elements(std::vector<element>::it
     if(error) {
       continue;
     }
-    accumulator.reserve(accumulator.size() + child_result.size());
     accumulator.insert(accumulator.end(),
                         std::make_move_iterator(child_result.begin()),
                         std::make_move_iterator(child_result.end()));
@@ -9003,6 +9992,12 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<dom::element>:
   if (error()) { return error(); }
   return first.get_string();
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<dom::element>::get_u8string() const noexcept {
+  if (error()) { return error(); }
+  return first.get_u8string();
+}
+#endif
 simdjson_inline simdjson_result<int64_t> simdjson_result<dom::element>::get_int64() const noexcept {
   if (error()) { return error(); }
   return first.get_int64();
@@ -9069,6 +10064,7 @@ simdjson_inline simdjson_result<dom::element> simdjson_result<dom::element>::at_
   return first.at_pointer(json_pointer);
 }
 simdjson_inline simdjson_result<dom::element> simdjson_result<dom::element>::at_path(const std::string_view json_path) const noexcept {
+  if (error()) { return error(); }
   auto json_pointer = json_path_to_pointer_conversion(json_path);
   if (json_pointer == "-1") { return INVALID_JSON_POINTER; }
   return at_pointer(json_pointer);
@@ -9201,6 +10197,13 @@ inline simdjson_result<std::string_view> element::get_string() const noexcept {
       return INCORRECT_TYPE;
   }
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+inline simdjson_result<std::u8string_view> element::get_u8string() const noexcept {
+  std::string_view v;
+  SIMDJSON_TRY(get_string().get(v));
+  return internal::as_u8string_view(v);
+}
+#endif
 inline simdjson_result<uint64_t> element::get_uint64() const noexcept {
   SIMDJSON_DEVELOPMENT_ASSERT(tape.usable()); // https://github.com/simdjson/simdjson/issues/1914
   if(simdjson_unlikely(!tape.is_uint64())) { // branch rarely taken
@@ -9296,6 +10299,9 @@ template<> inline simdjson_result<array> element::get<array>() const noexcept {
 template<> inline simdjson_result<object> element::get<object>() const noexcept { return get_object(); }
 template<> inline simdjson_result<const char *> element::get<const char *>() const noexcept { return get_c_str(); }
 template<> inline simdjson_result<std::string_view> element::get<std::string_view>() const noexcept { return get_string(); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> inline simdjson_result<std::u8string_view> element::get<std::u8string_view>() const noexcept { return get_u8string(); }
+#endif
 template<> inline simdjson_result<int64_t> element::get<int64_t>() const noexcept { return get_int64(); }
 template<> inline simdjson_result<uint64_t> element::get<uint64_t>() const noexcept { return get_uint64(); }
 template<> inline simdjson_result<double> element::get<double>() const noexcept { return get_double(); }
@@ -9628,6 +10634,29 @@ inline simdjson_result<element> parser::parse_into_document(document& provided_d
   return provided_doc.root();
 }

+inline simdjson_result<element> parser::parse_into_document_unpadded(document& provided_doc, const uint8_t *buf, size_t len) & noexcept {
+  // Like parse_into_document with realloc_if_needed=false (no copy, parse in
+  // place), but we tell the implementation the buffer is not padded so stage 2
+  // avoids reading past buf+len: string unescaping is bounded, near-the-end
+  // numbers are parsed from a padded copy, and atoms use length-aware
+  // validators (see tape_builder). Stage 1 is already safe for unpadded input.
+  error_code _error = ensure_capacity(provided_doc, len);
+  if (_error) { return _error; }
+
+  if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+    buf += 3;
+    len -= 3;
+  }
+  implementation->_number_as_string = _number_as_string;
+  implementation->_unpadded = true;
+  _error = implementation->parse(buf, len, provided_doc);
+  implementation->_unpadded = false; // restore so later padded parses use the fast path
+
+  if (_error) { return _error; }
+
+  return provided_doc.root();
+}
+
 simdjson_inline simdjson_result<element> parser::parse_into_document(document& provided_doc, const char *buf, size_t len, bool realloc_if_needed) & noexcept {
   return parse_into_document(provided_doc, reinterpret_cast<const uint8_t *>(buf), len, realloc_if_needed);
 }
@@ -9656,13 +10685,18 @@ simdjson_inline simdjson_result<element> parser::parse(const padded_string_view
   return parse(v.data(), v.length(), false);
 }

+inline simdjson_result<element> parser::parse_unpadded(const uint8_t *buf, size_t len) & noexcept {
+  return parse_into_document_unpadded(doc, buf, len);
+}
+simdjson_inline simdjson_result<element> parser::parse_unpadded(const char *buf, size_t len) & noexcept {
+  return parse_unpadded(reinterpret_cast<const uint8_t *>(buf), len);
+}
+simdjson_inline simdjson_result<element> parser::parse_unpadded(std::string_view s) & noexcept {
+  return parse_unpadded(reinterpret_cast<const uint8_t *>(s.data()), s.size());
+}
+
 inline simdjson_result<document_stream> parser::parse_many(const uint8_t *buf, size_t len, size_t batch_size) noexcept {
-  if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
-  if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
-    buf += 3;
-    len -= 3;
-  }
-  return document_stream(*this, buf, len, batch_size);
+  return parse_many(buf, len, batch_size, stream_format::whitespace_delimited);
 }
 inline simdjson_result<document_stream> parser::parse_many(const char *buf, size_t len, size_t batch_size) noexcept {
   return parse_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size);
@@ -9673,6 +10707,48 @@ inline simdjson_result<document_stream> parser::parse_many(const std::string &s,
 inline simdjson_result<document_stream> parser::parse_many(const padded_string &s, size_t batch_size) noexcept {
   return parse_many(s.data(), s.length(), batch_size);
 }
+inline simdjson_result<document_stream> parser::parse_many(const padded_string_view &v, size_t batch_size) noexcept {
+  return parse_many(v.data(), v.length(), batch_size);
+}
+
+inline simdjson_result<document_stream> parser::parse_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+  if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
+  if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+    buf += 3;
+    len -= 3;
+  }
+  if (format == stream_format::comma_delimited_array) {
+    // Strip leading JSON whitespace.
+    while (len > 0 && (buf[0] == ' ' || buf[0] == '\t' || buf[0] == '\n' || buf[0] == '\r')) {
+      buf++; len--;
+    }
+    // Expect the opening '['.
+    if (len == 0 || buf[0] != '[') { return TAPE_ERROR; }
+    buf++; len--;
+    // Strip trailing JSON whitespace.
+    while (len > 0 && (buf[len-1] == ' ' || buf[len-1] == '\t' || buf[len-1] == '\n' || buf[len-1] == '\r')) {
+      len--;
+    }
+    // Expect the closing ']'.
+    if (len == 0 || buf[len-1] != ']') { return TAPE_ERROR; }
+    len--;
+    // Fall through to comma_delimited over the array contents.
+    format = stream_format::comma_delimited;
+  }
+  return document_stream(*this, buf, len, batch_size, format);
+}
+inline simdjson_result<document_stream> parser::parse_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+  return parse_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size, format);
+}
+inline simdjson_result<document_stream> parser::parse_many(const std::string &s, size_t batch_size, stream_format format) noexcept {
+  return parse_many(s.data(), s.length(), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::parse_many(const padded_string &s, size_t batch_size, stream_format format) noexcept {
+  return parse_many(s.data(), s.length(), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::parse_many(const padded_string_view &v, size_t batch_size, stream_format format) noexcept {
+  return parse_many(v.data(), v.length(), batch_size, format);
+}

 simdjson_inline size_t parser::capacity() const noexcept {
   return implementation ? implementation->capacity() : 0;
@@ -9830,12 +10906,14 @@ simdjson_inline document_stream::document_stream(
   dom::parser &_parser,
   const uint8_t *_buf,
   size_t _len,
-  size_t _batch_size
+  size_t _batch_size,
+  stream_format _format
 ) noexcept
   : parser{&_parser},
     buf{_buf},
     len{_len},
     batch_size{_batch_size <= MINIMAL_BATCH_SIZE ? MINIMAL_BATCH_SIZE : _batch_size},
+    format{_format},
     error{SUCCESS}
 #ifdef SIMDJSON_THREADS_ENABLED
     , use_thread(_parser.threaded) // we need to make a copy because _parser.threaded can change
@@ -9853,6 +10931,7 @@ simdjson_inline document_stream::document_stream() noexcept
     buf{nullptr},
     len{0},
     batch_size{0},
+    format{stream_format::whitespace_delimited},
     error{UNINITIALIZED}
 #ifdef SIMDJSON_THREADS_ENABLED
     , use_thread(false)
@@ -9929,6 +11008,7 @@ inline void document_stream::start() noexcept {
   if (error) { return; }
   error = parser->ensure_capacity(batch_size);
   if (error) { return; }
+  parser->implementation->_number_as_string = parser->number_as_string();
   // Always run the first stage 1 parse immediately
   batch_start = 0;
   error = run_stage1(*parser, batch_start);
@@ -9968,7 +11048,40 @@ simdjson_inline std::string_view document_stream::iterator::source() const noexc
   } else {
     size_t next_doc_index = stream->batch_start + stream->parser->implementation->structural_indexes[stream->parser->implementation->next_structural_index];
     size_t svlen = next_doc_index - current_index();
-    while(svlen > 1 && (std::isspace(start[svlen-1]) || start[svlen-1] == '\0')) {
+    // When the scalar is followed by a truncated document, the structural
+    // indexes of that document were dropped and next_doc_index is the end of
+    // the input, so we bound the scalar by scanning the token itself.
+    size_t token_len = 0;
+    if (*start == '"') {
+      token_len = 1;
+      while (token_len < svlen) {
+        char c = start[token_len++];
+        if (c == '\\') {
+          token_len++;
+        } else if (c == '"') {
+          break;
+        }
+      }
+    } else {
+      while (token_len < svlen) {
+        char c = start[token_len];
+        if (std::isspace(static_cast<unsigned char>(c)) || c == ',' || c == '{' || c == '[' || c == '\0' || static_cast<uint8_t>(c) == 0x1E) {
+          break;
+        }
+        token_len++;
+      }
+    }
+    if (token_len > 0 && token_len < svlen) {
+      svlen = token_len;
+    }
+    // Trim trailing whitespace, NUL, and RS (0x1E). In RFC 7464 json_sequence
+    // mode the scanner classifies RS as a scalar character, so an RS-prefixed
+    // scalar document (number/true/false/null/string) has no closing structural
+    // index and the slice runs all the way up to the next document's RS. RS
+    // cannot legally appear in a JSON value at the source level (control
+    // characters in strings must be escaped as \u001E), so stripping it is
+    // safe in every stream_format.
+    while(svlen > 1 && (std::isspace(static_cast<unsigned char>(start[svlen-1])) || start[svlen-1] == '\0' || static_cast<uint8_t>(start[svlen-1]) == 0x1E || (stream->format == stream_format::comma_delimited && start[svlen-1] == ','))) {
       svlen--;
     }
     return std::string_view(start, svlen);
@@ -10008,6 +11121,9 @@ inline size_t document_stream::size_in_bytes() const noexcept {
 }

 inline size_t document_stream::truncated_bytes() const noexcept {
+  // Stage 1 returns EMPTY on zero-length input before it writes the index
+  // sentinels read below, so they would still hold a previous stream's values.
+  if (len == 0) { return 0; }
   if(error == CAPACITY) { return len - batch_start; }
   return parser->implementation->structural_indexes[parser->implementation->n_structural_indexes] - parser->implementation->structural_indexes[parser->implementation->n_structural_indexes + 1];
 }
@@ -10018,10 +11134,35 @@ inline size_t document_stream::next_batch_start() const noexcept {

 inline error_code document_stream::run_stage1(dom::parser &p, size_t _batch_start) noexcept {
   size_t remaining = len - _batch_start;
+  stage1_mode mode;
   if (remaining <= batch_size) {
-    return p.implementation->stage1(&buf[_batch_start], remaining, stage1_mode::streaming_final);
+    // Final batch
+    switch (format) {
+      case stream_format::json_sequence:
+        mode = stage1_mode::json_sequence_final;
+        break;
+      case stream_format::comma_delimited:
+        mode = stage1_mode::comma_delimited_final;
+        break;
+      default:
+        mode = stage1_mode::streaming_final;
+        break;
+    }
+    return p.implementation->stage1(&buf[_batch_start], remaining, mode);
   } else {
-    return p.implementation->stage1(&buf[_batch_start], batch_size, stage1_mode::streaming_partial);
+    // Partial batch
+    switch (format) {
+      case stream_format::json_sequence:
+        mode = stage1_mode::json_sequence_partial;
+        break;
+      case stream_format::comma_delimited:
+        mode = stage1_mode::comma_delimited_partial;
+        break;
+      default:
+        mode = stage1_mode::streaming_partial;
+        break;
+    }
+    return p.implementation->stage1(&buf[_batch_start], batch_size, mode);
   }
 }

@@ -10194,16 +11335,25 @@ inline error_code document::allocate(size_t capacity) noexcept {
     allocated_capacity = 0;
     return SUCCESS;
   }
+  if (capacity > SIMDJSON_MAXSIZE_BYTES) {
+    return CAPACITY;
+  }

   // a pathological input like "[[[[..." would generate capacity tape elements, so
   // need a capacity of at least capacity + 1, but it is also possible to do
   // worse with "[7,7,7,7,6,7,7,7,6,7,7,6,[7,7,7,7,6,7,7,7,6,7,7,6,7,7,7,7,7,7,6"
   //where capacity + 1 tape elements are
   // generated, see issue https://github.com/simdjson/simdjson/issues/345
+  if(capacity + 3 < capacity) {
+    return CAPACITY; // overflow, only happen on legacy 32-bit systems with very large capacity
+  }
   size_t tape_capacity = SIMDJSON_ROUNDUP_N(capacity + 3, 64);
   // a document with only zero-length strings... could have capacity/3 string
   // and we would need capacity/3 * 5 bytes on the string buffer
-  size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * capacity / 3 + SIMDJSON_PADDING, 64);
+  if(5 * (capacity / 3) + SIMDJSON_PADDING < SIMDJSON_PADDING) {
+    return CAPACITY; // overflow, only happen on legacy 32-bit systems with very large capacity
+  }
+  size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * (capacity / 3) + SIMDJSON_PADDING, 64);
   string_buf.reset( new (std::nothrow) uint8_t[string_capacity]);
   tape.reset(new (std::nothrow) uint64_t[tape_capacity]);
   if(!(string_buf && tape)) {
@@ -10347,6 +11497,7 @@ inline bool document::dump_raw_tape(std::ostream &os) const noexcept {
 /* skipped duplicate #include "simdjson/dom/object-inl.h" */
 /* skipped duplicate #include "simdjson/internal/tape_ref-inl.h" */

+#include <cmath>
 #include <cstring>

 namespace simdjson {
@@ -10517,11 +11668,29 @@ simdjson_inline void base_formatter<formatter>::number(int64_t x) {

 template <class formatter>
 simdjson_inline void base_formatter<formatter>::number(double x) {
-  char number_buffer[24];
+#if SIMDJSON_ENABLE_NAN_INF
+  if (simdjson_unlikely(!std::isfinite(x))) {
+    if (std::isnan(x)) {
+      char const *s = "NaN";
+      chars(s, s + 3);
+    } else {
+      if (x < 0) {
+        one_char('-');
+      }
+      char const *s = "Infinity";
+      chars(s, s + 8);
+    }
+    return;
+  }
+#endif
+
+  // Must be to_chars_buffer_size (40): only ~24 chars are emitted, but
+  // to_chars over-writes with fixed-size 16/17-byte copies for inlining.
+  char number_buffer[simdjson::internal::to_chars_buffer_size];
   // Currently, passing the nullptr to the second argument is
   // safe because our implementation does not check the second
   // argument.
-  char *newp = internal::to_chars(number_buffer, nullptr, x);
+  char *newp = simdjson::internal::to_chars(number_buffer, nullptr, x);
   chars(number_buffer, newp);
 }

@@ -10797,7 +11966,7 @@ inline void string_builder<serializer>::append(simdjson::dom::element value) {
       format.string(iter.get_string_view());
       break;
     case tape_type::BIGINT: {
-      // Big integer stored as string — output raw digits (no quotes)
+      // Big integer stored as string -- output raw digits (no quotes)
       auto sv = iter.get_string_view();
       format.chars(sv.data(), sv.data() + sv.size());
       break;
@@ -10935,7 +12104,7 @@ simdjson_inline std::string_view string_builder<serializer>::str() const {
 #include <vector>
 #include <string>
 #include <string_view>
-#include <set>
+#include <utility>

 namespace simdjson {
 namespace internal {
@@ -10944,12 +12113,50 @@ namespace internal {
  * Layout mode for fractured JSON formatting.
  */
 enum class layout_mode {
-  INLINE,              // Single line: [1, 2, 3] or {"a": 1}
-  COMPACT_MULTILINE,   // Multiple items per line with breaks
-  TABLE,               // Tabular format for arrays of similar objects
-  EXPANDED             // Traditional multi-line with indentation
+  single_line,         // Single line: [1, 2, 3] or {"a": 1}
+  compact_multiline,   // Multiple items per line with breaks
+  table,               // Tabular format for arrays of similar objects
+  expanded             // Traditional multi-line with indentation
 };

+/** Kind of value found in a table column across all rows that have one. */
+enum class table_column_type {
+  unknown,
+  simple,   // string, bool, or null
+  number,
+  array,
+  object,
+  mixed     // rows disagree on kind
+};
+
+/** Column of a table-formatted array.*/
+struct table_column {
+  /** Column name for object rows (may be the empty string). */
+  std::string key{};
+  /** True for object rows (the column has a key), false for array rows. */
+  bool has_key = false;
+  /** Rendered length of key */
+  size_t key_width = 0;
+  table_column_type type = table_column_type::unknown;
+  /** Widest rendered value in this column (when fully expanding all children) */
+  size_t width = 0;
+  /** Widest plain value in this column. */
+  size_t plain_width = 0;
+  /** subcolumns (only populated if every child is an array or every child is an object) */
+  std::vector<table_column> children{};
+};
+
+/** Whether rows from this column list use nested vs. simple bracket padding
+ * One shared decision, since picking it per-row would misalign width-aligned rows. */
+inline bool table_row_is_nested(const std::vector<table_column>& columns) {
+  for (const table_column& col : columns) {
+    if (col.type == table_column_type::object || col.type == table_column_type::array) {
+      return true;
+    }
+  }
+  return false;
+}
+
 /**
  * Metrics computed for a JSON element during structure analysis.
  * These metrics drive layout decisions and contain child metrics for recursive formatting.
@@ -10970,16 +12177,32 @@ struct element_metrics {
   /** Is this an array where all elements have similar structure? */
   bool is_uniform_array = false;

-  /** For uniform arrays of objects: the common keys */
-  std::vector<std::string> common_keys{};
-
-  /** Recommended layout mode based on analysis */
-  layout_mode recommended_layout = layout_mode::EXPANDED;
+  /** for uniform arrays: this array's columns for alignment */
+  std::vector<table_column> table_columns{};
+  /** Widest table row after pruning recursive columns that don't fit into line budget */
+  size_t table_row_width = 0;
+  /** Widest table row without pruning recursive columns that don't fit into line budget */
+  size_t table_row_width_full = 0;

   /** Child metrics for arrays and objects (in order of iteration) */
   std::vector<element_metrics> children{};
+
+  /** For scalar uniform arrays (table_columns empty): the rows' common type. */
+  table_column_type scalar_column_type = table_column_type::unknown;
 };

+/** children[idx], or a default-constructed element_metrics if idx is out of range */
+inline const element_metrics& child_metrics_at(const std::vector<element_metrics>& children, size_t idx) {
+  static const element_metrics empty{};
+  return idx < children.size() ? children[idx] : empty;
+}
+
+/** *ptr, or a default-constructed element_metrics if ptr is null. */
+inline const element_metrics& child_metrics_at(const element_metrics* ptr) {
+  static const element_metrics empty{};
+  return ptr ? *ptr : empty;
+}
+
 /**
  * Analyzes JSON structure to compute metrics for formatting decisions.
  *
@@ -11040,20 +12263,27 @@ public:
   element_metrics analyze_object(const dom::object& obj,
                                  const fractured_json_options& opts);

+  /** Decide layout at the given render depth. Kept out of analysis since
+   * the same metrics can render inline or expanded at different depths. */
+  static layout_mode decide_layout(const element_metrics& metrics,
+                                    size_t depth,
+                                    const fractured_json_options& opts,
+                                    bool has_trailing_comma = false);
+
 private:
   const fractured_json_options* current_opts_ = nullptr;

   /** Recursive analysis implementation */
-  element_metrics analyze_element(const dom::element& elem, size_t depth);
+  element_metrics analyze_element(const dom::element& elem, size_t depth) const;

   /** Analyze scalar values (strings, numbers, booleans, null) */
-  element_metrics analyze_scalar(const dom::element& elem);
+  element_metrics analyze_scalar(const dom::element& elem) const;

   /** Analyze an array element */
-  element_metrics analyze_array(const dom::array& arr, size_t depth);
+  element_metrics analyze_array(const dom::array& arr, size_t depth) const;

   /** Analyze an object element */
-  element_metrics analyze_object(const dom::object& obj, size_t depth);
+  element_metrics analyze_object(const dom::object& obj, size_t depth) const;

   /** Estimate inline length for a string (including quotes and escaping) */
   size_t estimate_string_length(std::string_view s) const;
@@ -11064,27 +12294,35 @@ private:
   size_t estimate_number_length(uint64_t u) const;

   /**
-   * Check if an array contains uniform objects suitable for table formatting.
+   * Check if an array contains uniformly-shaped rows suitable for table
+   * formatting, filling in corresponding metrics
    * @param arr The array to check
-   * @param common_keys Output: keys common to all objects
-   * @return true if the array is suitable for table formatting
+   * @param metrics The array's metrics, with children already filled in
+   * @param depth The array's depth
    */
-  bool check_array_uniformity(const dom::array& arr,
-                               std::vector<std::string>& common_keys) const;
+  bool check_array_uniformity(const dom::array& arr, element_metrics& metrics, size_t depth) const;

-  /**
-   * Compute similarity between two objects.
-   * @return Fraction of keys that are common (0.0 to 1.0)
-   */
-  double compute_object_similarity(const dom::object& a,
-                                   const dom::object& b) const;
+  static table_column_type classify_table_value(dom::element_type type);

-  /**
-   * Decide the recommended layout mode based on metrics and options.
-   */
-  layout_mode decide_layout(const element_metrics& metrics,
-                            size_t depth,
-                            size_t available_width) const;
+  /** Find common type across a set of sibling values and the widest of their rendered lengths */
+  static void classify_and_measure(const std::vector<std::pair<dom::element, const element_metrics*>>& values,
+                                    table_column_type& common, size_t& max_width);
+
+  /** Recursively build table columns for an array */
+  void build_table_columns(const std::vector<std::pair<dom::element, const element_metrics*>>& values,
+                            std::vector<table_column>& out_columns) const;
+
+  /** Rendered width of a row, assuming each column's current */
+  size_t compute_columns_width(const std::vector<table_column>& columns) const;
+
+  /** Height of a column's recursion (0 = leaf). */
+  static size_t column_height(const table_column& column);
+
+  /** Flatten deepest columns of the table */
+  static bool flatten_deepest_columns(std::vector<table_column>& columns);
+
+  /** Bottom-up refresh of column widths after flatten_deepest_columns. */
+  void recompute_column_widths(std::vector<table_column>& columns) const;
 };

 } // namespace internal
@@ -11130,56 +12368,35 @@ public:
   /** Get the current layout mode */
   layout_mode get_layout_mode() const;

-  /** Set current depth for formatting decisions */
-  void set_depth(size_t depth);
-
-  /** Get current depth */
-  size_t get_depth() const;
-
   /** Track current line length for compact multiline decisions */
   void track_line_length(size_t chars);

-  /** Reset line length (after newline) */
-  void reset_line_length();
-
-  /** Get current line length */
-  size_t get_line_length() const;
-
   /** Check if we should break to a new line in compact mode */
   bool should_break_line(size_t upcoming_length) const;

   /** Get the options */
   const fractured_json_options& options() const;

-  // Table formatting support
-  /** Begin a table row */
-  void begin_table_row();
-
-  /** End a table row */
-  void end_table_row();
-
-  /** Set column widths for table alignment */
-  void set_column_widths(const std::vector<size_t>& widths);
-
-  /** Get current column index in table mode */
-  size_t get_column_index() const;
-
-  /** Advance to next column */
-  void next_column();
-
-  /** Add padding to align with column width */
-  void align_to_column_width(size_t actual_width);
-
 private:
   fractured_json_options options_;
-  layout_mode current_layout_ = layout_mode::EXPANDED;
-  size_t current_depth_ = 0;
+  layout_mode current_layout_ = layout_mode::expanded;
   size_t current_line_length_ = 0;
+};

-  // Table state
-  bool in_table_mode_ = false;
-  std::vector<size_t> column_widths_;
-  size_t current_column_ = 0;
+/** RAII helper forcing single line layout and restoring previous mode on exit */
+class scoped_single_line_mode {
+public:
+  explicit scoped_single_line_mode(fractured_formatter& format)
+      : format_(format), prev_(format.get_layout_mode()) {
+    format_.set_layout_mode(layout_mode::single_line);
+  }
+  ~scoped_single_line_mode() { format_.set_layout_mode(prev_); }
+  scoped_single_line_mode(const scoped_single_line_mode&) = delete;
+  scoped_single_line_mode& operator=(const scoped_single_line_mode&) = delete;
+
+private:
+  fractured_formatter& format_;
+  layout_mode prev_;
 };

 /**
@@ -11214,10 +12431,12 @@ private:
   fractured_json_options options_;

   /** Format an element using pre-computed metrics */
-  void format_element(const dom::element& elem, const element_metrics& metrics, size_t depth);
+  void format_element(const dom::element& elem, const element_metrics& metrics, size_t depth,
+                       bool has_trailing_comma = false);

   /** Format an array with the appropriate layout */
-  void format_array(const dom::array& arr, const element_metrics& metrics, size_t depth);
+  void format_array(const dom::array& arr, const element_metrics& metrics, size_t depth,
+                     bool has_trailing_comma = false);

   /** Format an array inline: [1, 2, 3] */
   void format_array_inline(const dom::array& arr, const element_metrics& metrics);
@@ -11225,14 +12444,51 @@ private:
   /** Format an array with compact multiline: multiple items per line */
   void format_array_compact_multiline(const dom::array& arr, const element_metrics& metrics, size_t depth);

+  /** Like format_array_compact_multiline, but rows are cross-row aligned
+   * and packed using a fixed per-row slot width. */
+  void format_array_compact_multiline_aligned(const dom::array& arr, const element_metrics& metrics, size_t depth);
+
   /** Format an array as a table */
   void format_array_as_table(const dom::array& arr, const element_metrics& metrics, size_t depth);

+  /** Write one object row's columns */
+  void format_table_object_row(const dom::object& obj, const element_metrics& row_metrics,
+                                const std::vector<table_column>& columns, size_t depth);
+
+  /** Write one array row's columns */
+  void format_table_array_row(const dom::array& arr, const element_metrics& row_metrics,
+                               const std::vector<table_column>& columns, size_t depth);
+
+  /** Dispatches to format_table_object_row/format_table_array_row based on elem's type. */
+  void format_table_row(const dom::element& elem, const element_metrics& row_metrics,
+                        const std::vector<table_column>& columns, size_t depth);
+
+  /** Row for a uniform scalar array: writes elem inline, then pads to width so every row lines up. */
+  void format_table_scalar_row(const dom::element& elem, const element_metrics& row_metrics,
+                                size_t width, size_t depth, table_column_type column_type);
+
+  /** Shared per-column writer: recurses if the column has children,
+   * otherwise writes a plain padded value or blank. */
+  void format_table_row_columns(const std::vector<table_column>& columns,
+                                 const std::vector<bool>& found,
+                                 const std::vector<dom::element>& values,
+                                 const std::vector<const element_metrics*>& value_metrics,
+                                 size_t depth);
+
+  /** Writes a single aligned leaf value */
+  void format_table_leaf_value(const dom::element& elem, const element_metrics& vm, size_t width,
+                                table_column_type column_type, bool needs_comma,
+                                bool add_comma_space, size_t depth);
+
+  /** Whether, for a column of the given type, the comma goes right after the value */
+  bool comma_goes_before_padding(table_column_type column_type) const;
+
   /** Format an array expanded: one item per line */
   void format_array_expanded(const dom::array& arr, const element_metrics& metrics, size_t depth);

   /** Format an object with the appropriate layout */
-  void format_object(const dom::object& obj, const element_metrics& metrics, size_t depth);
+  void format_object(const dom::object& obj, const element_metrics& metrics, size_t depth,
+                      bool has_trailing_comma = false);

   /** Format an object inline: {"a": 1, "b": 2} */
   void format_object_inline(const dom::object& obj, const element_metrics& metrics);
@@ -11243,12 +12499,8 @@ private:
   /** Format a scalar value */
   void format_scalar(const dom::element& elem);

-  /** Calculate column widths for table formatting */
-  std::vector<size_t> calculate_column_widths(const dom::array& arr,
-                                               const std::vector<std::string>& columns) const;
-
-  /** Measure the actual formatted length of a value (for alignment) */
-  size_t measure_value_length(const dom::element& elem) const;
+  /** Whether to pad this container's own brackets. */
+  bool bracket_padding_for(const element_metrics& metrics) const;
 };

 } // namespace internal
@@ -11260,6 +12512,8 @@ private:
 #include <cmath>
 #include <algorithm>
 #include <cstring>
+#include <iterator>
+#include <unordered_map>

 namespace simdjson {
 namespace internal {
@@ -11290,7 +12544,7 @@ inline element_metrics structure_analyzer::analyze_object(const dom::object& obj
   return analyze_object(obj, 0);
 }

-inline element_metrics structure_analyzer::analyze_element(const dom::element& elem, size_t depth) {
+inline element_metrics structure_analyzer::analyze_element(const dom::element& elem, size_t depth) const {
   switch (elem.type()) {
     case dom::element_type::ARRAY: {
       dom::array arr;
@@ -11313,12 +12567,11 @@ inline element_metrics structure_analyzer::analyze_element(const dom::element& e
   return element_metrics{};
 }

-inline element_metrics structure_analyzer::analyze_scalar(const dom::element& elem) {
+inline element_metrics structure_analyzer::analyze_scalar(const dom::element& elem) const {
   element_metrics metrics;
   metrics.complexity = 0;
   metrics.child_count = 0;
   metrics.can_inline = true;
-  metrics.recommended_layout = layout_mode::INLINE;

   switch (elem.type()) {
     case dom::element_type::STRING: {
@@ -11367,7 +12620,7 @@ inline element_metrics structure_analyzer::analyze_scalar(const dom::element& el
 }

 inline element_metrics structure_analyzer::analyze_array(const dom::array& arr,
-                                                          size_t depth) {
+                                                          size_t depth) const {
   element_metrics metrics;
   metrics.complexity = 1; // At least 1 for being an array
   metrics.estimated_inline_len = 2; // "[]"
@@ -11392,35 +12645,31 @@ inline element_metrics structure_analyzer::analyze_array(const dom::array& arr,
   // Complexity is 1 + max child complexity
   metrics.complexity = 1 + max_child_complexity;

-  // Check if can inline
-  metrics.can_inline = (metrics.complexity <= current_opts_->max_inline_complexity) &&
-                       (metrics.estimated_inline_len <= current_opts_->max_inline_length);
-
-  // Check for uniform array (table formatting)
-  if (current_opts_->enable_table_format &&
-      metrics.child_count >= current_opts_->min_table_rows) {
-    metrics.is_uniform_array = check_array_uniformity(arr, metrics.common_keys);
+  // Bracket padding "[ 1, 2 ]" vs "[1, 2]"
+  bool use_bracket_padding = (max_child_complexity >= 1)
+      ? current_opts_->nested_bracket_padding : current_opts_->simple_bracket_padding;
+  if (use_bracket_padding && metrics.child_count > 0) {
+    metrics.estimated_inline_len += 2;
   }

-  // Decide layout
-  if (metrics.child_count == 0) {
-    metrics.recommended_layout = layout_mode::INLINE;
-  } else if (metrics.can_inline) {
-    metrics.recommended_layout = layout_mode::INLINE;
-  } else if (metrics.is_uniform_array && !metrics.common_keys.empty()) {
-    metrics.recommended_layout = layout_mode::TABLE;
-  } else if (current_opts_->enable_compact_multiline &&
-             max_child_complexity <= current_opts_->max_compact_array_complexity) {
-    metrics.recommended_layout = layout_mode::COMPACT_MULTILINE;
-  } else {
-    metrics.recommended_layout = layout_mode::EXPANDED;
+  // Check if can inline
+  metrics.can_inline = metrics.complexity <= current_opts_->max_inline_complexity;
+
+  // Check for uniform array (table formatting, or aligned compact multiline).
+  bool wants_table_columns =
+      (current_opts_->enable_table_format &&
+       max_child_complexity <= current_opts_->max_table_row_complexity) ||
+      (current_opts_->enable_compact_multiline &&
+       max_child_complexity <= current_opts_->max_compact_array_complexity);
+  if (wants_table_columns) {
+    metrics.is_uniform_array = check_array_uniformity(arr, metrics, depth);
   }

   return metrics;
 }

 inline element_metrics structure_analyzer::analyze_object(const dom::object& obj,
-                                                           size_t depth) {
+                                                           size_t depth) const {
   element_metrics metrics;
   metrics.complexity = 1;
   metrics.estimated_inline_len = 2; // "{}"
@@ -11447,16 +12696,15 @@ inline element_metrics structure_analyzer::analyze_object(const dom::object& obj

   metrics.complexity = 1 + max_child_complexity;

-  metrics.can_inline = (metrics.complexity <= current_opts_->max_inline_complexity) &&
-                       (metrics.estimated_inline_len <= current_opts_->max_inline_length);
-
-  // Objects use inline or expanded (no table/compact for objects)
-  if (metrics.child_count == 0 || metrics.can_inline) {
-    metrics.recommended_layout = layout_mode::INLINE;
-  } else {
-    metrics.recommended_layout = layout_mode::EXPANDED;
+  // Bracket padding '{ "a": 1 }' vs '{"a": 1}'
+  bool use_bracket_padding = (max_child_complexity >= 1)
+      ? current_opts_->nested_bracket_padding : current_opts_->simple_bracket_padding;
+  if (use_bracket_padding && metrics.child_count > 0) {
+    metrics.estimated_inline_len += 2;
   }

+  metrics.can_inline = metrics.complexity <= current_opts_->max_inline_complexity;
+
   return metrics;
 }

@@ -11473,8 +12721,18 @@ inline size_t structure_analyzer::estimate_string_length(std::string_view s) con
 }

 inline size_t structure_analyzer::estimate_number_length(double d) const {
-  if (std::isnan(d) || std::isinf(d)) {
+  if (!std::isfinite(d)) {
+#if SIMDJSON_ENABLE_NAN_INF
+    if (std::isnan(d)) {
+      return 3; // "NaN"
+    } else if (d < 0) {
+      return 9; // "-Infinity"
+    } else {
+      return 8; // "Infinity"
+    }
+#else
     return 4; // "null" for invalid numbers
+#endif
   }
   // Rough estimate: up to 17 significant digits + sign + decimal point + exponent
   char buf[32];
@@ -11505,115 +12763,290 @@ inline size_t structure_analyzer::estimate_number_length(uint64_t u) const {
   return len;
 }

-inline bool structure_analyzer::check_array_uniformity(const dom::array& arr,
-                                                        std::vector<std::string>& common_keys) const {
-  common_keys.clear();
-
-  std::set<std::string> shared_keys;
-  dom::object first_obj;
-  bool have_first = false;
-  size_t object_count = 0;
+inline table_column_type structure_analyzer::classify_table_value(dom::element_type type) {
+  switch (type) {
+    case dom::element_type::OBJECT: return table_column_type::object;
+    case dom::element_type::ARRAY: return table_column_type::array;
+    case dom::element_type::INT64:
+    case dom::element_type::UINT64:
+    case dom::element_type::DOUBLE: return table_column_type::number;
+    case dom::element_type::NULL_VALUE: return table_column_type::unknown;
+    default: return table_column_type::simple; // string, bool
+  }
+}

-  for (dom::element elem : arr) {
-    if (elem.type() != dom::element_type::OBJECT) {
-      return false; // Not all elements are objects
+inline void structure_analyzer::classify_and_measure(
+    const std::vector<std::pair<dom::element, const element_metrics*>>& values,
+    table_column_type& common, size_t& max_width) {
+  common = table_column_type::unknown;
+  max_width = 0;
+  for (const auto& v : values) {
+    table_column_type t = classify_table_value(v.first.type());
+    if (t != table_column_type::unknown) {
+      if (common == table_column_type::unknown) common = t;
+      else if (t != common) common = table_column_type::mixed;
     }
-
-    dom::object obj;
-    if (elem.get_object().get(obj) != SUCCESS) {
-      return false;
+    if (v.second) {
+      max_width = (std::max)(max_width, v.second->estimated_inline_len);
     }
+  }
+}

-    std::set<std::string> current_keys;
-    for (dom::key_value_pair field : obj) {
-      current_keys.insert(std::string(field.key));
+inline void structure_analyzer::build_table_columns(
+    const std::vector<std::pair<dom::element, const element_metrics*>>& values,
+    std::vector<table_column>& out_columns) const {
+  out_columns.clear();
+  if (values.empty()) {
+    return;
+  }
+
+  table_column_type common = table_column_type::unknown;
+  for (const auto& v : values) {
+    table_column_type t = classify_table_value(v.first.type());
+    if (t == table_column_type::unknown) continue;
+    if (common == table_column_type::unknown) common = t;
+    else if (t != common) { common = table_column_type::mixed; break; }
+  }
+  if (common != table_column_type::object && common != table_column_type::array) {
+    return;
+  }
+
+  std::vector<std::vector<std::pair<dom::element, const element_metrics*>>> per_column_values;
+
+  if (common == table_column_type::object) {
+    std::unordered_map<std::string_view, size_t> column_index;
+    // Last row (1-based) that filled each column, to detect duplicate keys.
+    std::vector<size_t> column_last_row;
+    size_t row = 0;
+    for (const auto& v : values) {
+      if (v.first.type() != dom::element_type::OBJECT) continue;
+      dom::object obj;
+      if (v.first.get_object().get(obj) != SUCCESS) continue;
+
+      row++;
+      size_t field_idx = 0;
+      for (dom::key_value_pair field : obj) {
+        auto it = column_index.find(field.key);
+        size_t col_idx;
+        if (it == column_index.end()) {
+          col_idx = out_columns.size();
+          column_index.emplace(field.key, col_idx);
+          out_columns.emplace_back();
+          out_columns.back().key.assign(field.key.data(), field.key.size());
+          out_columns.back().has_key = true;
+          out_columns.back().key_width = estimate_string_length(field.key);
+          per_column_values.emplace_back();
+          column_last_row.push_back(0);
+        } else {
+          col_idx = it->second;
+        }
+        // A row with a duplicate key cannot be laid out as a table: each
+        // column holds one value per row, so the later duplicates would be
+        // lost. Returning no columns also disables the aligned compact
+        // multiline layout for this array, which needs the same columns.
+        if (column_last_row[col_idx] == row) {
+          out_columns.clear();
+          return;
+        }
+        column_last_row[col_idx] = row;
+        const element_metrics* field_metrics = (v.second && field_idx < v.second->children.size())
+            ? &v.second->children[field_idx] : nullptr;
+        per_column_values[col_idx].emplace_back(field.value, field_metrics);
+        field_idx++;
+      }
     }
+  } else { // array: columns by position
+    for (const auto& v : values) {
+      if (v.first.type() != dom::element_type::ARRAY) continue;
+      dom::array sub_arr;
+      if (v.first.get_array().get(sub_arr) != SUCCESS) continue;

-    if (!have_first) {
-      shared_keys = current_keys;
-      first_obj = obj;
-      have_first = true;
-    } else {
-      // Check similarity threshold against the first object
-      double similarity = compute_object_similarity(first_obj, obj);
-      if (similarity < current_opts_->table_similarity_threshold) {
-        return false; // Objects are too dissimilar for table format
+      size_t idx = 0;
+      for (dom::element item : sub_arr) {
+        if (out_columns.size() <= idx) {
+          out_columns.emplace_back();
+          per_column_values.emplace_back();
+        }
+        const element_metrics* item_metrics = (v.second && idx < v.second->children.size())
+            ? &v.second->children[idx] : nullptr;
+        per_column_values[idx].emplace_back(item, item_metrics);
+        idx++;
       }
+    }
+  }
+
+  for (size_t i = 0; i < out_columns.size(); i++) {
+    table_column_type col_type;
+    size_t max_width;
+    classify_and_measure(per_column_values[i], col_type, max_width);
+    out_columns[i].type = col_type;
+    out_columns[i].plain_width = max_width;

-      // Intersect with current keys
-      std::set<std::string> intersection;
-      std::set_intersection(shared_keys.begin(), shared_keys.end(),
-                            current_keys.begin(), current_keys.end(),
-                            std::inserter(intersection, intersection.begin()));
-      shared_keys = intersection;
+    if (col_type == table_column_type::object || col_type == table_column_type::array) {
+      build_table_columns(per_column_values[i], out_columns[i].children);
     }

-    object_count++;
+    out_columns[i].width = out_columns[i].children.empty() ? max_width : compute_columns_width(out_columns[i].children);
   }
+}

-  if (object_count < current_opts_->min_table_rows) {
-    return false;
+inline bool structure_analyzer::check_array_uniformity(const dom::array& arr,
+                                                        element_metrics& metrics,
+                                                        size_t depth) const {
+  std::vector<std::pair<dom::element, const element_metrics*>> values;
+  values.reserve(metrics.child_count);
+
+  size_t row_idx = 0;
+  for (dom::element elem : arr) {
+    const element_metrics* row_metrics = (row_idx < metrics.children.size()) ? &metrics.children[row_idx] : nullptr;
+    values.emplace_back(elem, row_metrics);
+    row_idx++;
+  }
+
+  build_table_columns(values, metrics.table_columns);
+  if (metrics.table_columns.empty()) {
+    // Not uniformly object or array. Check for uniform scalar
+    table_column_type common;
+    size_t max_width;
+    classify_and_measure(values, common, max_width);
+    if (common != table_column_type::number && common != table_column_type::simple) {
+      return false;
+    }
+    metrics.scalar_column_type = common;
+    metrics.table_row_width = max_width;
+    metrics.table_row_width_full = max_width;
+    return true;
   }

-  // Require at least one common key for table formatting
-  if (shared_keys.empty()) {
-    return false;
+  metrics.table_row_width_full = compute_columns_width(metrics.table_columns);
+
+  size_t row_indent_width = (depth + 1) * current_opts_->indent_spaces;
+  size_t budget = (row_indent_width + 1 >= current_opts_->max_total_line_length)
+      ? 0 : current_opts_->max_total_line_length - row_indent_width - 1;
+
+  size_t width = metrics.table_row_width_full;
+  while (width > budget && flatten_deepest_columns(metrics.table_columns)) {
+    recompute_column_widths(metrics.table_columns);
+    width = compute_columns_width(metrics.table_columns);
   }

-  common_keys.assign(shared_keys.begin(), shared_keys.end());
+  metrics.table_row_width = width;
   return true;
 }

-inline double structure_analyzer::compute_object_similarity(const dom::object& a,
-                                                             const dom::object& b) const {
-  std::set<std::string> keys_a, keys_b;
-  for (dom::key_value_pair field : a) {
-    keys_a.insert(std::string(field.key));
+inline size_t structure_analyzer::compute_columns_width(const std::vector<table_column>& columns) const {
+  size_t width = 2; // "{}" or "[]"
+  if (table_row_is_nested(columns) ? current_opts_->nested_bracket_padding : current_opts_->simple_bracket_padding) {
+    width += 2;
   }
-  for (dom::key_value_pair field : b) {
-    keys_b.insert(std::string(field.key));
+
+  for (const table_column& col : columns) {
+    if (col.has_key) {
+      width += col.key_width;
+      width += current_opts_->colon_padding ? 2 : 1;
+    }
+    width += col.width;
+  }
+  if (columns.size() > 1) {
+    width += (columns.size() - 1) * (current_opts_->comma_padding ? 2 : 1);
+  }
+  return width;
+}
+
+inline size_t structure_analyzer::column_height(const table_column& column) {
+  size_t height = 0;
+  for (const table_column& child : column.children) {
+    height = (std::max)(height, column_height(child));
   }
+  return column.children.empty() ? 0 : height + 1;
+}

-  std::set<std::string> intersection;
-  std::set_intersection(keys_a.begin(), keys_a.end(),
-                        keys_b.begin(), keys_b.end(),
-                        std::inserter(intersection, intersection.begin()));
+inline bool structure_analyzer::flatten_deepest_columns(std::vector<table_column>& columns) {
+  size_t max_height = 0;
+  for (const table_column& col : columns) {
+    max_height = (std::max)(max_height, column_height(col));
+  }

-  std::set<std::string> union_set;
-  std::set_union(keys_a.begin(), keys_a.end(),
-                 keys_b.begin(), keys_b.end(),
-                 std::inserter(union_set, union_set.begin()));
+  bool changed = false;
+  for (table_column& col : columns) {
+    if (column_height(col) != max_height || max_height == 0) continue;
+    if (max_height == 1) {
+      col.children.clear();
+      col.width = col.plain_width;
+      changed = true;
+    } else {
+      changed |= flatten_deepest_columns(col.children);
+    }
+  }
+  return changed;
+}

-  if (union_set.empty()) return 1.0;
-  return static_cast<double>(intersection.size()) / static_cast<double>(union_set.size());
+inline void structure_analyzer::recompute_column_widths(std::vector<table_column>& columns) const {
+  for (table_column& col : columns) {
+    if (!col.children.empty()) {
+      recompute_column_widths(col.children);
+      col.width = compute_columns_width(col.children);
+    }
+  }
 }

 inline layout_mode structure_analyzer::decide_layout(const element_metrics& metrics,
                                                       size_t depth,
-                                                      size_t available_width) const {
+                                                      const fractured_json_options& opts,
+                                                      bool has_trailing_comma) {
   if (metrics.child_count == 0) {
-    return layout_mode::INLINE;
+    return layout_mode::single_line;
   }

+  long long signed_depth = static_cast<long long>(depth);
+  bool depth_allows_inline_or_compact = signed_depth > opts.always_expand_depth;
+  bool depth_allows_table = signed_depth >= opts.always_expand_depth;
+
   // Check inline feasibility
-  size_t indent_width = depth * current_opts_->indent_spaces;
-  if (metrics.can_inline &&
-      metrics.estimated_inline_len + indent_width <= available_width) {
-    return layout_mode::INLINE;
+  size_t reserved_width = depth * opts.indent_spaces + (has_trailing_comma ? 1 : 0);
+  if (depth_allows_inline_or_compact && metrics.can_inline &&
+      metrics.estimated_inline_len + reserved_width <= opts.max_total_line_length) {
+    return layout_mode::single_line;
   }

-  // Check table mode
-  if (metrics.is_uniform_array && !metrics.common_keys.empty()) {
-    return layout_mode::TABLE;
-  }
+  // Rows (table's or compact multiline's) render one level deeper than the
+  // array itself.
+  size_t row_indent_width = (depth + 1) * opts.indent_spaces;

   // Check compact multiline
-  if (current_opts_->enable_compact_multiline &&
-      metrics.complexity <= current_opts_->max_compact_array_complexity + 1) {
-    return layout_mode::COMPACT_MULTILINE;
+  // for uniform arrays fall back to table if we would have to flatten any formatting
+  bool compact_multiline_enabled = opts.enable_compact_multiline &&
+      metrics.complexity <= opts.max_compact_array_complexity + 1 &&
+      metrics.child_count >= opts.min_compact_array_row_items;
+  if (depth_allows_inline_or_compact && compact_multiline_enabled) {
+    bool aligned = metrics.is_uniform_array;
+    size_t comma_width = opts.comma_padding ? 2 : 1;
+    size_t avg_item_width;
+    if (aligned) {
+      avg_item_width = metrics.table_row_width_full + comma_width;
+    } else {
+      size_t sum = 0;
+      for (const element_metrics& child : metrics.children) {
+        sum += child.estimated_inline_len;
+      }
+      avg_item_width = comma_width + sum / metrics.child_count;
+    }
+
+    size_t row_pack_space = (row_indent_width >= opts.max_total_line_length)
+        ? 0 : opts.max_total_line_length - row_indent_width;
+    if (avg_item_width * opts.min_compact_array_row_items <= row_pack_space) {
+      return layout_mode::compact_multiline;
+    }
+  }
+
+  // Check Table mode
+  if (depth_allows_table && opts.enable_table_format &&
+      metrics.is_uniform_array &&
+      metrics.table_row_width + 1 + row_indent_width <= opts.max_total_line_length) {
+    return layout_mode::table;
   }

-  return layout_mode::EXPANDED;
+  return layout_mode::expanded;
 }

 //
@@ -11621,10 +13054,10 @@ inline layout_mode structure_analyzer::decide_layout(const element_metrics& metr
 //

 inline fractured_formatter::fractured_formatter(const fractured_json_options& opts)
-    : options_(opts), column_widths_{} {}
+    : options_(opts) {}

 simdjson_inline void fractured_formatter::print_newline() {
-  if (current_layout_ == layout_mode::INLINE) {
+  if (current_layout_ == layout_mode::single_line) {
     return; // No newlines in inline mode
   }
   one_char('\n');
@@ -11632,7 +13065,7 @@ simdjson_inline void fractured_formatter::print_newline() {
 }

 simdjson_inline void fractured_formatter::print_indents(size_t depth) {
-  if (current_layout_ == layout_mode::INLINE) {
+  if (current_layout_ == layout_mode::single_line) {
     return; // No indentation in inline mode
   }
   for (size_t i = 0; i < depth * options_.indent_spaces; i++) {
@@ -11654,26 +13087,10 @@ inline layout_mode fractured_formatter::get_layout_mode() const {
   return current_layout_;
 }

-inline void fractured_formatter::set_depth(size_t depth) {
-  current_depth_ = depth;
-}
-
-inline size_t fractured_formatter::get_depth() const {
-  return current_depth_;
-}
-
 inline void fractured_formatter::track_line_length(size_t chars) {
   current_line_length_ += chars;
 }

-inline void fractured_formatter::reset_line_length() {
-  current_line_length_ = 0;
-}
-
-inline size_t fractured_formatter::get_line_length() const {
-  return current_line_length_;
-}
-
 inline bool fractured_formatter::should_break_line(size_t upcoming_length) const {
   return (current_line_length_ + upcoming_length) > options_.max_total_line_length;
 }
@@ -11682,39 +13099,6 @@ inline const fractured_json_options& fractured_formatter::options() const {
   return options_;
 }

-inline void fractured_formatter::begin_table_row() {
-  in_table_mode_ = true;
-  current_column_ = 0;
-}
-
-inline void fractured_formatter::end_table_row() {
-  in_table_mode_ = false;
-  current_column_ = 0;
-}
-
-inline void fractured_formatter::set_column_widths(const std::vector<size_t>& widths) {
-  column_widths_ = widths;
-}
-
-inline size_t fractured_formatter::get_column_index() const {
-  return current_column_;
-}
-
-inline void fractured_formatter::next_column() {
-  current_column_++;
-}
-
-inline void fractured_formatter::align_to_column_width(size_t actual_width) {
-  if (current_column_ < column_widths_.size()) {
-    size_t target_width = column_widths_[current_column_];
-    while (actual_width < target_width) {
-      one_char(' ');
-      actual_width++;
-      current_line_length_++;
-    }
-  }
-}
-
 //
 // Fractured String Builder Implementation
 //
@@ -11753,19 +13137,20 @@ simdjson_inline std::string_view fractured_string_builder::str() const {

 inline void fractured_string_builder::format_element(const dom::element& elem,
                                                        const element_metrics& metrics,
-                                                       size_t depth) {
+                                                       size_t depth,
+                                                       bool has_trailing_comma) {
   switch (elem.type()) {
     case dom::element_type::ARRAY: {
       dom::array arr;
       if (elem.get_array().get(arr) == SUCCESS) {
-        format_array(arr, metrics, depth);
+        format_array(arr, metrics, depth, has_trailing_comma);
       }
       break;
     }
     case dom::element_type::OBJECT: {
       dom::object obj;
       if (elem.get_object().get(obj) == SUCCESS) {
-        format_object(obj, metrics, depth);
+        format_object(obj, metrics, depth, has_trailing_comma);
       }
       break;
     }
@@ -11777,18 +13162,20 @@ inline void fractured_string_builder::format_element(const dom::element& elem,

 inline void fractured_string_builder::format_array(const dom::array& arr,
                                                     const element_metrics& metrics,
-                                                    size_t depth) {
-  switch (metrics.recommended_layout) {
-    case layout_mode::INLINE:
+                                                    size_t depth,
+                                                    bool has_trailing_comma) {
+  layout_mode layout = structure_analyzer::decide_layout(metrics, depth, options_, has_trailing_comma);
+  switch (layout) {
+    case layout_mode::single_line:
       format_array_inline(arr, metrics);
       break;
-    case layout_mode::COMPACT_MULTILINE:
+    case layout_mode::compact_multiline:
       format_array_compact_multiline(arr, metrics, depth);
       break;
-    case layout_mode::TABLE:
+    case layout_mode::table:
       format_array_as_table(arr, metrics, depth);
       break;
-    case layout_mode::EXPANDED:
+    case layout_mode::expanded:
     default:
       format_array_expanded(arr, metrics, depth);
       break;
@@ -11797,8 +13184,7 @@ inline void fractured_string_builder::format_array(const dom::array& arr,

 inline void fractured_string_builder::format_array_inline(const dom::array& arr,
                                                             const element_metrics& metrics) {
-  layout_mode prev_layout = format_.get_layout_mode();
-  format_.set_layout_mode(layout_mode::INLINE);
+  scoped_single_line_mode single_line(format_);

   format_.start_array();

@@ -11812,60 +13198,69 @@ inline void fractured_string_builder::format_array_inline(const dom::array& arr,
       if (options_.comma_padding) {
         format_.print_space();
       }
-    } else if (options_.simple_bracket_padding) {
+    } else if (bracket_padding_for(metrics)) {
       format_.print_space();
     }
     first = false;
-    const element_metrics& child_metrics = (child_idx < metrics.children.size())
-        ? metrics.children[child_idx] : element_metrics{};
+    const element_metrics& child_metrics = child_metrics_at(metrics.children, child_idx);
     format_element(elem, child_metrics, 0);
     child_idx++;
   }

-  if (options_.simple_bracket_padding && !empty) {
+  if (bracket_padding_for(metrics) && !empty) {
     format_.print_space();
   }
   format_.end_array();
-
-  format_.set_layout_mode(prev_layout);
 }

 inline void fractured_string_builder::format_array_compact_multiline(const dom::array& arr,
                                                                        const element_metrics& metrics,
                                                                        size_t depth) {
+  if (metrics.is_uniform_array) {
+    format_array_compact_multiline_aligned(arr, metrics, depth);
+    return;
+  }
+
   format_.start_array();
   format_.print_newline();
   format_.print_indents(depth + 1);

-  size_t items_on_line = 0;
   bool first = true;
+  bool prev_item_was_expanded = false;
   size_t child_idx = 0;

   for (dom::element elem : arr) {
+    const element_metrics& child_metrics = child_metrics_at(metrics.children, child_idx);
+
     if (!first) {
       format_.comma();
+      format_.track_line_length(1);

       // Check if we should break to new line
-      if (items_on_line >= options_.max_items_per_line ||
-          format_.should_break_line(20)) { // 20 is rough estimate for next item
+      if (prev_item_was_expanded ||
+          format_.should_break_line(child_metrics.estimated_inline_len)) {
         format_.print_newline();
         format_.print_indents(depth + 1);
-        items_on_line = 0;
       } else if (options_.comma_padding) {
         format_.print_space();
       }
     }
     first = false;

-    // Format element inline
-    layout_mode prev_layout = format_.get_layout_mode();
-    format_.set_layout_mode(layout_mode::INLINE);
-    const element_metrics& child_metrics = (child_idx < metrics.children.size())
-        ? metrics.children[child_idx] : element_metrics{};
-    format_element(elem, child_metrics, depth + 1);
-    format_.set_layout_mode(prev_layout);
+    bool is_last = (child_idx + 1 == metrics.child_count);
+    layout_mode item_layout = structure_analyzer::decide_layout(child_metrics, depth + 1, options_, !is_last);
+    bool item_fits = item_layout == layout_mode::single_line;
+    if (item_fits) {
+      {
+        scoped_single_line_mode single_line(format_);
+        format_element(elem, child_metrics, depth + 1, !is_last);
+      }
+      format_.track_line_length(child_metrics.estimated_inline_len);
+    } else {
+      format_element(elem, child_metrics, depth + 1, !is_last);
+    }
+    prev_item_was_expanded = !item_fits;

-    items_on_line++;
     child_idx++;
   }

@@ -11874,114 +13269,309 @@ inline void fractured_string_builder::format_array_compact_multiline(const dom::
   format_.end_array();
 }

-inline void fractured_string_builder::format_array_as_table(const dom::array& arr,
-                                                             const element_metrics& metrics,
-                                                             size_t depth) {
-  const std::vector<std::string>& columns = metrics.common_keys;
-  if (columns.empty()) {
-    format_array_expanded(arr, metrics, depth);
-    return;
-  }
-
-  // Calculate column widths for alignment
-  std::vector<size_t> col_widths = calculate_column_widths(arr, columns);
-  format_.set_column_widths(col_widths);
+inline void fractured_string_builder::format_array_compact_multiline_aligned(
+    const dom::array& arr, const element_metrics& metrics, size_t depth) {
+  const std::vector<table_column>& columns = metrics.table_columns;

   format_.start_array();
   format_.print_newline();
+  format_.print_indents(depth + 1);

-  bool first_row = true;
+  size_t indent_width = (depth + 1) * options_.indent_spaces;
+  size_t available_line_space = (indent_width >= options_.max_total_line_length)
+      ? 0 : options_.max_total_line_length - indent_width;
+  size_t comma_width = options_.comma_padding ? 2 : 1;
+  size_t remaining_line_space = available_line_space;
+
+  bool first = true;
   size_t child_idx = 0;
+
   for (dom::element elem : arr) {
-    if (!first_row) {
-      format_.comma();
-      format_.print_newline();
-    }
-    first_row = false;
+    bool needs_comma = (child_idx + 1 < metrics.child_count);
+    size_t space_needed = metrics.table_row_width_full + (needs_comma ? comma_width : 0);

-    format_.print_indents(depth + 1);
-    format_.begin_table_row();
+    if (!first) {
+      if (remaining_line_space < space_needed) {
+        format_.print_newline();
+        format_.print_indents(depth + 1);
+        remaining_line_space = available_line_space;
+      } else if (options_.comma_padding) {
+        format_.print_space();
+      }
+    }
+    first = false;

-    // Format object as inline with aligned columns
-    dom::object obj;
-    if (elem.get_object().get(obj) != SUCCESS) {
-      child_idx++;
-      continue;
+    const element_metrics& row_metrics = child_metrics_at(metrics.children, child_idx);
+    if (columns.empty()) {
+      format_table_scalar_row(elem, row_metrics, metrics.table_row_width_full, depth + 1,
+                              metrics.scalar_column_type);
+    } else {
+      format_table_row(elem, row_metrics, columns, depth + 1);
     }
+    if (needs_comma) {
+      format_.comma();
+    }
+    remaining_line_space -= (std::min)(remaining_line_space, space_needed);
+    child_idx++;
+  }

-    // Get child metrics for this row (object)
-    const element_metrics& row_metrics = (child_idx < metrics.children.size())
-        ? metrics.children[child_idx] : element_metrics{};
+  format_.print_newline();
+  format_.print_indents(depth);
+  format_.end_array();
+}

-    format_.start_object();
-    if (options_.simple_bracket_padding) {
-      format_.print_space();
-    }
+inline void fractured_string_builder::format_table_row_columns(
+    const std::vector<table_column>& columns,
+    const std::vector<bool>& found,
+    const std::vector<dom::element>& values,
+    const std::vector<const element_metrics*>& value_metrics,
+    size_t depth) {
+  const size_t num_columns = columns.size();
+  size_t last_present_idx = num_columns;
+  for (size_t i = 0; i < num_columns; i++) {
+    if (found[i]) last_present_idx = i;
+  }

-    bool first_col = true;
-    const size_t num_columns = columns.size();
+  size_t comma_width = options_.comma_padding ? 2 : 1;

-    for (size_t col_idx = 0; col_idx < num_columns; col_idx++) {
-      const std::string& key = columns[col_idx];
-      const bool is_last_col = (col_idx == num_columns - 1);
+  for (size_t col_idx = 0; col_idx < num_columns; col_idx++) {
+    const table_column& column = columns[col_idx];
+    const bool is_last_col = (col_idx == num_columns - 1);

-      if (!first_col) {
-        format_.comma();
-        if (options_.comma_padding) {
+    if (found[col_idx]) {
+      if (column.has_key) {
+        format_.key(column.key);
+        if (options_.colon_padding) {
           format_.print_space();
         }
       }
-      first_col = false;

-      // Write key
-      format_.key(key);
-      if (options_.colon_padding) {
-        format_.print_space();
-      }
+      bool needs_comma = !is_last_col && (col_idx < last_present_idx);

-      // Find the value for this key and its metrics
-      dom::element value;
-      bool found = false;
-      size_t field_idx = 0;
-      for (dom::key_value_pair field : obj) {
-        if (field.key == key) {
-          value = field.value;
-          found = true;
-          break;
+      if (!column.children.empty()) {
+        // Recurses into this cell's own columns instead of a plain value;
+        // every row aligns those the same way (blank-padding missing
+        // ones), so the result is always exactly column.width wide
+        // no padding needed afterward, unlike the leaf case below.
+        if (column.type == table_column_type::object) {
+          dom::object sub_obj;
+          if (values[col_idx].get_object().get(sub_obj) == SUCCESS) {
+            const element_metrics& sub_metrics = child_metrics_at(value_metrics[col_idx]);
+            format_table_object_row(sub_obj, sub_metrics, column.children, depth);
+          }
+        } else {
+          dom::array sub_arr;
+          if (values[col_idx].get_array().get(sub_arr) == SUCCESS) {
+            const element_metrics& sub_metrics = child_metrics_at(value_metrics[col_idx]);
+            format_table_array_row(sub_arr, sub_metrics, column.children, depth);
+          }
         }
-        field_idx++;
+        // value is already padded
+        if (needs_comma) {
+          format_.comma();
+          if (options_.comma_padding) {
+            format_.print_space();
+          }
+        }
+      } else {
+        const element_metrics& vm = child_metrics_at(value_metrics[col_idx]);
+        format_table_leaf_value(values[col_idx], vm, column.width, column.type, needs_comma,
+                                 /*add_comma_space=*/true, depth);
       }

-      // Write value
-      if (found) {
-        layout_mode prev_layout = format_.get_layout_mode();
-        format_.set_layout_mode(layout_mode::INLINE);
-        const element_metrics& value_metrics = (field_idx < row_metrics.children.size())
-            ? row_metrics.children[field_idx] : element_metrics{};
-        format_element(value, value_metrics, depth + 1);
-        format_.set_layout_mode(prev_layout);
-      } else {
-        format_.null_atom();
+      if (!is_last_col && !needs_comma) {
+        // Found, but no more real values follow: blank space where a comma would go.
+        for (size_t i = 0; i < comma_width; i++) {
+          format_.one_char(' ');
+        }
+      }
+    } else {
+      size_t slot_width = column.width;
+      if (column.has_key) {
+        slot_width += column.key_width + (options_.colon_padding ? 2 : 1);
+      }
+      for (size_t i = 0; i < slot_width; i++) {
+        format_.one_char(' ');
       }

-      // Only pad non-last columns to align values across rows
       if (!is_last_col) {
-        size_t actual_len = found ? measure_value_length(value) : 4; // 4 for "null"
-        size_t target_width = col_widths[col_idx];
-        while (actual_len < target_width) {
+        for (size_t i = 0; i < comma_width; i++) {
           format_.one_char(' ');
-          actual_len++;
         }
       }
+    }
+  }
+}

-      format_.next_column();
+inline void fractured_string_builder::format_table_object_row(
+    const dom::object& obj, const element_metrics& row_metrics,
+    const std::vector<table_column>& columns, size_t depth) {
+  const size_t num_columns = columns.size();
+  std::vector<bool> found(num_columns, false);
+  std::vector<dom::element> values(num_columns);
+  std::vector<const element_metrics*> value_metrics(num_columns, nullptr);
+
+  for (size_t col_idx = 0; col_idx < num_columns; col_idx++) {
+    size_t field_idx = 0;
+    for (dom::key_value_pair field : obj) {
+      if (field.key == columns[col_idx].key) {
+        found[col_idx] = true;
+        values[col_idx] = field.value;
+        value_metrics[col_idx] = (field_idx < row_metrics.children.size())
+            ? &row_metrics.children[field_idx] : nullptr;
+        break;
+      }
+      field_idx++;
     }
+  }

-    if (options_.simple_bracket_padding) {
-      format_.print_space();
+  bool nested = table_row_is_nested(columns);
+  format_.start_object();
+  if (nested ? options_.nested_bracket_padding : options_.simple_bracket_padding) {
+    format_.print_space();
+  }
+  format_table_row_columns(columns, found, values, value_metrics, depth);
+  if (nested ? options_.nested_bracket_padding : options_.simple_bracket_padding) {
+    format_.print_space();
+  }
+  format_.end_object();
+}
+
+inline void fractured_string_builder::format_table_array_row(
+    const dom::array& arr, const element_metrics& row_metrics,
+    const std::vector<table_column>& columns, size_t depth) {
+  const size_t num_columns = columns.size();
+  std::vector<bool> found(num_columns, false);
+  std::vector<dom::element> values(num_columns);
+  std::vector<const element_metrics*> value_metrics(num_columns, nullptr);
+
+  size_t idx = 0;
+  for (dom::element item : arr) {
+    if (idx >= num_columns) break;
+    found[idx] = true;
+    values[idx] = item;
+    value_metrics[idx] = (idx < row_metrics.children.size()) ? &row_metrics.children[idx] : nullptr;
+    idx++;
+  }
+
+  bool nested = table_row_is_nested(columns);
+  format_.start_array();
+  if (nested ? options_.nested_bracket_padding : options_.simple_bracket_padding) {
+    format_.print_space();
+  }
+  format_table_row_columns(columns, found, values, value_metrics, depth);
+  if (nested ? options_.nested_bracket_padding : options_.simple_bracket_padding) {
+    format_.print_space();
+  }
+  format_.end_array();
+}
+
+inline void fractured_string_builder::format_table_row(
+    const dom::element& elem, const element_metrics& row_metrics,
+    const std::vector<table_column>& columns, size_t depth) {
+  if (elem.type() == dom::element_type::ARRAY) {
+    dom::array arr;
+    if (elem.get_array().get(arr) == SUCCESS) {
+      format_table_array_row(arr, row_metrics, columns, depth);
+    }
+  } else {
+    dom::object obj;
+    if (elem.get_object().get(obj) == SUCCESS) {
+      format_table_object_row(obj, row_metrics, columns, depth);
+    }
+  }
+}
+
+inline bool fractured_string_builder::comma_goes_before_padding(table_column_type column_type) const {
+  switch (options_.comma_placement) {
+    case table_comma_placement::before_padding: return true;
+    case table_comma_placement::after_padding: return false;
+    case table_comma_placement::before_padding_except_numbers:
+    default:
+      return column_type != table_column_type::number;
+  }
+}
+
+inline void fractured_string_builder::format_table_leaf_value(
+    const dom::element& elem, const element_metrics& vm, size_t width,
+    table_column_type column_type, bool needs_comma, bool add_comma_space, size_t depth) {
+  bool comma_before_pad = needs_comma && comma_goes_before_padding(column_type);
+  bool comma_after_pad = needs_comma && !comma_before_pad;
+
+  bool right_align = column_type == table_column_type::number &&
+      options_.number_alignment == number_list_alignment::right;
+
+  size_t value_len = vm.estimated_inline_len;
+  size_t left_pad = 0;
+  size_t right_pad = 0;
+  if (right_align) {
+    left_pad = (width > value_len) ? width - value_len : 0;
+    comma_before_pad = needs_comma;
+    comma_after_pad = false;
+  } else {
+    right_pad = (width > value_len) ? width - value_len : 0;
+  }
+
+  for (size_t i = 0; i < left_pad; i++) {
+    format_.one_char(' ');
+  }
+
+  {
+    scoped_single_line_mode single_line(format_);
+    format_element(elem, vm, depth);
+  }
+
+  if (comma_before_pad) {
+    format_.comma();
+  }
+  for (size_t i = 0; i < right_pad; i++) {
+    format_.one_char(' ');
+  }
+  if (comma_after_pad) {
+    format_.comma();
+  }
+  if (needs_comma && add_comma_space && options_.comma_padding) {
+    format_.print_space();
+  }
+}
+
+inline void fractured_string_builder::format_table_scalar_row(
+    const dom::element& elem, const element_metrics& row_metrics, size_t width, size_t depth,
+    table_column_type column_type) {
+  format_table_leaf_value(elem, row_metrics, width, column_type,
+                          /*needs_comma=*/false, /*add_comma_space=*/false, depth);
+}
+
+inline void fractured_string_builder::format_array_as_table(const dom::array& arr,
+                                                             const element_metrics& metrics,
+                                                             size_t depth) {
+  if (!metrics.is_uniform_array) {
+    format_array_expanded(arr, metrics, depth);
+    return;
+  }
+  const std::vector<table_column>& columns = metrics.table_columns;
+
+  format_.start_array();
+  format_.print_newline();
+
+  bool first_row = true;
+  size_t child_idx = 0;
+  for (dom::element elem : arr) {
+    if (!first_row) {
+      format_.comma();
+      format_.print_newline();
+    }
+    first_row = false;
+
+    format_.print_indents(depth + 1);
+
+    const element_metrics& row_metrics = child_metrics_at(metrics.children, child_idx);
+    if (columns.empty()) {
+      format_table_scalar_row(elem, row_metrics, metrics.table_row_width, depth + 1,
+                              metrics.scalar_column_type);
+    } else {
+      format_table_row(elem, row_metrics, columns, depth + 1);
     }
-    format_.end_object();
-    format_.end_table_row();
     child_idx++;
   }

@@ -12008,9 +13598,9 @@ inline void fractured_string_builder::format_array_expanded(const dom::array& ar

     format_.print_newline();
     format_.print_indents(depth + 1);
-    const element_metrics& child_metrics = (child_idx < metrics.children.size())
-        ? metrics.children[child_idx] : element_metrics{};
-    format_element(elem, child_metrics, depth + 1);
+    const element_metrics& child_metrics = child_metrics_at(metrics.children, child_idx);
+    bool is_last = (child_idx + 1 == metrics.child_count);
+    format_element(elem, child_metrics, depth + 1, !is_last);
     child_idx++;
   }

@@ -12023,8 +13613,10 @@ inline void fractured_string_builder::format_array_expanded(const dom::array& ar

 inline void fractured_string_builder::format_object(const dom::object& obj,
                                                      const element_metrics& metrics,
-                                                     size_t depth) {
-  if (metrics.recommended_layout == layout_mode::INLINE || metrics.can_inline) {
+                                                     size_t depth,
+                                                     bool has_trailing_comma) {
+  layout_mode layout = structure_analyzer::decide_layout(metrics, depth, options_, has_trailing_comma);
+  if (layout == layout_mode::single_line) {
     format_object_inline(obj, metrics);
   } else {
     format_object_expanded(obj, metrics, depth);
@@ -12033,8 +13625,7 @@ inline void fractured_string_builder::format_object(const dom::object& obj,

 inline void fractured_string_builder::format_object_inline(const dom::object& obj,
                                                              const element_metrics& metrics) {
-  layout_mode prev_layout = format_.get_layout_mode();
-  format_.set_layout_mode(layout_mode::INLINE);
+  scoped_single_line_mode single_line(format_);

   format_.start_object();

@@ -12049,7 +13640,7 @@ inline void fractured_string_builder::format_object_inline(const dom::object& ob
       if (options_.comma_padding) {
         format_.print_space();
       }
-    } else if (options_.simple_bracket_padding) {
+    } else if (bracket_padding_for(metrics)) {
       format_.print_space();
     }
     first = false;
@@ -12058,18 +13649,15 @@ inline void fractured_string_builder::format_object_inline(const dom::object& ob
     if (options_.colon_padding) {
       format_.print_space();
     }
-    const element_metrics& child_metrics = (child_idx < metrics.children.size())
-        ? metrics.children[child_idx] : element_metrics{};
+    const element_metrics& child_metrics = child_metrics_at(metrics.children, child_idx);
     format_element(field.value, child_metrics, 0);
     child_idx++;
   }

-  if (options_.simple_bracket_padding && !empty) {
+  if (bracket_padding_for(metrics) && !empty) {
     format_.print_space();
   }
   format_.end_object();
-
-  format_.set_layout_mode(prev_layout);
 }

 inline void fractured_string_builder::format_object_expanded(const dom::object& obj,
@@ -12094,9 +13682,9 @@ inline void fractured_string_builder::format_object_expanded(const dom::object&
     if (options_.colon_padding) {
       format_.print_space();
     }
-    const element_metrics& child_metrics = (child_idx < metrics.children.size())
-        ? metrics.children[child_idx] : element_metrics{};
-    format_element(field.value, child_metrics, depth + 1);
+    const element_metrics& child_metrics = child_metrics_at(metrics.children, child_idx);
+    bool is_last = (child_idx + 1 == metrics.child_count);
+    format_element(field.value, child_metrics, depth + 1, !is_last);
     child_idx++;
   }

@@ -12152,97 +13740,8 @@ inline void fractured_string_builder::format_scalar(const dom::element& elem) {
   }
 }

-inline size_t fractured_string_builder::measure_value_length(const dom::element& elem) const {
-  switch (elem.type()) {
-    case dom::element_type::STRING: {
-      std::string_view str;
-      if (elem.get_string().get(str) == SUCCESS) {
-        // Count actual escaped length
-        size_t len = 2; // quotes
-        for (char c : str) {
-          if (c == '"' || c == '\\' || static_cast<unsigned char>(c) < 32) {
-            len += 2; // escape sequence
-          } else {
-            len += 1;
-          }
-        }
-        return len;
-      }
-      return 2;
-    }
-    case dom::element_type::INT64: {
-      int64_t val;
-      if (elem.get_int64().get(val) == SUCCESS) {
-        if (val == 0) return 1;
-        // Handle INT64_MIN specially to avoid overflow when negating
-        if (val == INT64_MIN) return 20; // "-9223372036854775808" is 20 characters
-        size_t len = (val < 0) ? 1 : 0;
-        int64_t abs_val = (val < 0) ? -val : val;
-        while (abs_val > 0) { len++; abs_val /= 10; }
-        return len;
-      }
-      return 1;
-    }
-    case dom::element_type::UINT64: {
-      uint64_t val;
-      if (elem.get_uint64().get(val) == SUCCESS) {
-        if (val == 0) return 1;
-        size_t len = 0;
-        while (val > 0) { len++; val /= 10; }
-        return len;
-      }
-      return 1;
-    }
-    case dom::element_type::DOUBLE: {
-      double val;
-      if (elem.get_double().get(val) == SUCCESS) {
-        char buf[32];
-        int len = snprintf(buf, sizeof(buf), "%.17g", val);
-        return len > 0 ? static_cast<size_t>(len) : 1;
-      }
-      return 1;
-    }
-    case dom::element_type::BOOL: {
-      bool val;
-      if (elem.get_bool().get(val) == SUCCESS) {
-        return val ? 4 : 5; // "true" or "false"
-      }
-      return 5;
-    }
-    case dom::element_type::NULL_VALUE:
-      return 4; // "null"
-    default:
-      return 4;
-  }
-}
-
-inline std::vector<size_t> fractured_string_builder::calculate_column_widths(
-    const dom::array& arr,
-    const std::vector<std::string>& columns) const {
-
-  std::vector<size_t> widths(columns.size(), 0);
-
-  for (dom::element elem : arr) {
-    dom::object obj;
-    if (elem.get_object().get(obj) != SUCCESS) {
-      continue;
-    }
-
-    for (size_t col_idx = 0; col_idx < columns.size(); col_idx++) {
-      const std::string& key = columns[col_idx];
-
-      for (dom::key_value_pair field : obj) {
-        if (field.key == key) {
-          // Measure actual value length
-          size_t len = measure_value_length(field.value);
-          widths[col_idx] = (std::max)(widths[col_idx], len);
-          break;
-        }
-      }
-    }
-  }
-
-  return widths;
+inline bool fractured_string_builder::bracket_padding_for(const element_metrics& metrics) const {
+  return metrics.complexity >= 2 ? options_.nested_bracket_padding : options_.simple_bracket_padding;
 }

 } // namespace internal
@@ -12282,19 +13781,6 @@ std::string fractured_json(simdjson_result<T> x, const fractured_json_options& o
 }
 #endif

-// Explicit template instantiations for common types
-template std::string fractured_json(dom::element x);
-template std::string fractured_json(dom::element x, const fractured_json_options& options);
-template std::string fractured_json(dom::array x);
-template std::string fractured_json(dom::array x, const fractured_json_options& options);
-template std::string fractured_json(dom::object x);
-template std::string fractured_json(dom::object x, const fractured_json_options& options);
-
-#if SIMDJSON_EXCEPTIONS
-template std::string fractured_json(simdjson_result<dom::element> x);
-template std::string fractured_json(simdjson_result<dom::element> x, const fractured_json_options& options);
-#endif
-
 //
 // String-based API for formatting any JSON string
 //
@@ -12662,6 +14148,8 @@ enum instruction_set {
   LASX = 0x40000,
   //RVV = 0x80000,
   RVV_VLS = 0x100000,
+  SVE = 0x200000,
+  SVE2 = 0x400000,
 };

 } // namespace internal
@@ -13774,7 +15262,7 @@ public:
   simdjson_inline implementation() : simdjson::implementation(
       "rvv_vls",
       "RISC-V V extension",
-      0
+      internal::instruction_set::RVV_VLS
   ) {}
   simdjson_warn_unused error_code create_dom_parser_implementation(
     size_t capacity,
@@ -13933,7 +15421,14 @@ simdjson_inline int leading_zeroes(uint64_t input_num) {

 /* result might be undefined when input_num is zero */
 simdjson_inline int count_ones(uint64_t input_num) {
+#if SIMDJSON_REGULAR_VISUAL_STUDIO
    return vaddv_u8(vcnt_u8(vcreate_u8(input_num)));
+#else
+   // if the system supports SVE or CSSC, __builtin_popcountll
+   // might be compiled to fewer single instructions. For CSSC,
+   // __builtin_popcountll is compiled to a single instruction.
+   return __builtin_popcountll(input_num);
+#endif// SIMDJSON_REGULAR_VISUAL_STUDIO
 }


@@ -13970,15 +15465,6 @@ simdjson_inline uint64_t zero_leading_bit(uint64_t rev_bits, int leading_zeroes)

 #endif

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2, uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
-  *result = value1 + value2;
-  return *result < value1;
-#else
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-#endif
-}

 } // unnamed namespace
 } // namespace arm64
@@ -14242,6 +15728,7 @@ namespace {
       return vget_lane_u64(
           vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(*this), 4)), 0);
     }
+    // Integer umaxv, not a float compare: flush-to-zero would treat a denormal as zero.
     simdjson_inline bool any() const { return vmaxvq_u32(vreinterpretq_u32_u8(*this)) != 0; }
   };

@@ -14890,6 +16377,9 @@ namespace atomparsing {
 // to the compile-time constant 1936482662.
 simdjson_inline uint32_t string_to_uint32(const char* str) { uint32_t val; std::memcpy(&val, str, sizeof(uint32_t)); return val; }

+// Acts on the same principle as string_to_uint32, but on an 8-byte block of memory
+simdjson_inline uint64_t string_to_uint64(const char* str) { uint64_t val; std::memcpy(&val, str, sizeof(uint64_t)); return val; }
+

 // Again in str4ncmp we use a memcpy to avoid undefined behavior. The memcpy may appear expensive.
 // Yet all decent optimizing compilers will compile memcpy to a single instruction, just about.
@@ -14901,6 +16391,28 @@ simdjson_inline uint32_t str4ncmp(const uint8_t *src, const char* atom) {
   return srcval ^ string_to_uint32(atom);
 }

+// Checks that the first 8 characters of the input string match the given atom in a case-insensitive manner.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint64_t str8ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+  uint64_t srcval; // we want to avoid unaligned 32-bit loads (undefined in C/C++)
+  static_assert(sizeof(uint64_t) <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be larger than 8 bytes");
+  std::memcpy(&srcval, src, sizeof(uint64_t));
+
+  return (srcval | 0x2020202020202020ull) ^ string_to_uint64(atom);
+}
+
+// Checks that the first 3 characters of 'src' match 'atom' in a case-insensitive way.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint32_t str3ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+  return ((src[0] | 0x20) ^ atom[0]) //
+       | ((src[1] | 0x20) ^ atom[1]) //
+       | ((src[2] | 0x20) ^ atom[2]);
+}
+
 simdjson_warn_unused
 simdjson_inline bool is_valid_true_atom(const uint8_t *src) {
   return (str4ncmp(src, "true") | jsoncharutils::is_not_structural_or_whitespace(src[4])) == 0;
@@ -14937,6 +16449,71 @@ simdjson_inline bool is_valid_null_atom(const uint8_t *src, size_t len) {
   else { return false; }
 }

+#if SIMDJSON_ENABLE_NAN_INF
+// "nan" is 3 bytes; we check characters and then verify the next
+// character is structural or whitespace. We accept both "nan" and "NaN".
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src) {
+  return (str3ncmp_case_insensitive(src, "nan")
+        | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+}
+
+// checks that the next four characters of a string are 'nan"', where the 'nan'
+// is checked in a case-insensitive way.
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_in_string(const uint8_t *src) {
+  return (str3ncmp_case_insensitive(src, "nan") | (src[3] ^ '"')) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src, size_t len) {
+  if (len > 3) { return is_valid_nan_atom(src); }
+  if (len == 3) { return str3ncmp_case_insensitive(src, "nan") == 0; }
+  return false;
+}
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src) {
+  bool is_short_inf = (str3ncmp_case_insensitive(src, "inf")
+                     | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+  if(is_short_inf) return true;
+
+  // Check for 'infinity' (any capitalization)
+  return (str8ncmp_case_insensitive(src, "infinity") | jsoncharutils::is_not_structural_or_whitespace(src[8])) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_in_string(const uint8_t *src) {
+  bool is_short_inf = (str3ncmp_case_insensitive(src, "inf") | (src[3] ^ '"')) == 0;
+  if(is_short_inf) return true;
+
+  return (str8ncmp_case_insensitive(src, "infinity") | (src[8] ^ '"')) == 0;
+}
+
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src, size_t len) {
+  if (len > 8) { return is_valid_inf_atom(src); }
+  if (len == 8 && str8ncmp_case_insensitive(src, "infinity") == 0) {
+    return true;
+  }
+  if (len > 3) {
+    return (str3ncmp_case_insensitive(src, "inf")
+          | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+  }
+  if (len == 3) { return str3ncmp_case_insensitive(src, "inf") == 0; }
+  return false;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
 } // namespace atomparsing
 } // unnamed namespace
 } // namespace arm64
@@ -15027,7 +16604,7 @@ inline simdjson_warn_unused error_code dom_parser_implementation::set_capacity(s
 }

 inline simdjson_warn_unused error_code dom_parser_implementation::set_max_depth(size_t max_depth) noexcept {
-  if(max_depth == 0 || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
+  if(max_depth == 0 || max_depth > SIMDJSON_MAX_DEPTH || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
   // Stage 2 stacks
   open_containers.reset(new (std::nothrow) open_container[max_depth]);
   is_array.reset(new (std::nothrow) bool[max_depth]);
@@ -15206,6 +16783,7 @@ protected:
 /* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_NUMBERPARSING_H */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/base.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/jsoncharutils.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/atomparsing.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/internal/numberparsing_tables.h" */
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */

@@ -15245,6 +16823,18 @@ simdjson_inline double to_double(uint64_t mantissa, uint64_t real_exponent, bool
     return d;
 }

+// Convert a mantissa, an exponent and a sign bit into an ieee32 float (binary32).
+// The real_exponent needs to be in [0, 254] (technically real_exponent = 255 would be acceptable).
+// The mantissa should be in [0,1<<24). The bit at index (1U << 23) will be zeroed.
+simdjson_inline float to_float(uint32_t mantissa, uint32_t real_exponent, bool negative) {
+    float f;
+    mantissa &= ~(uint32_t(1) << 23);
+    mantissa |= real_exponent << 23;
+    mantissa |= ((static_cast<uint32_t>(negative)) << 31);
+    std::memcpy(&f, &mantissa, sizeof(f));
+    return f;
+}
+
 // Attempts to compute i * 10^(power) exactly; and if "negative" is
 // true, negate the result.
 // This function will only work in some cases, when it does not work, success is
@@ -15501,6 +17091,219 @@ simdjson_inline bool compute_float_64(int64_t power, uint64_t i, bool negative,
   return true;
 }

+// Attempts to compute i * 10^(power) as a binary32 (float) value; and if
+// "negative" is true, negate the result.
+//
+// This is the same Eisel-Lemire algorithm as compute_float_64, retargeted at
+// binary32. It is adapted from fast_float
+// (https://github.com/fastfloat/fast_float), where the routine is written once
+// and instantiated for each binary format. We only need 24 bits of mantissa
+// instead of 53, so we ask the truncated product for 24 + 2 = 26 accurate bits
+// (plus one bit that may be lost to the "upperbit" shift), and we shift the
+// product right by (upperbit + 64 - 23 - 3) instead of (upperbit + 64 - 52 - 3).
+//
+// The power-of-five table is shared with the binary64 code: the useful range
+// for binary32, [smallest_power_binary32, largest_power_binary32], is a strict
+// subset of [smallest_power, largest_power].
+//
+// The function returns false when the result would be infinite: simdjson
+// refuses to parse infinite values, so the caller reports an error. Unlike
+// compute_float_64, a false return never means "try harder": the accuracy of
+// the two-limb product is guaranteed by Noble Mushtak and Daniel Lemire,
+// "Fast Number Parsing Without Fallback" (https://arxiv.org/abs/2212.06644).
+// The caller must still fall back to a slow path when the 64-bit mantissa i
+// was truncated (more than 19 significant digits).
+//
+// We assume that power is in the [smallest_power, largest_power] interval: the
+// caller is responsible for this check.
+simdjson_inline bool compute_float_32(int64_t power, uint64_t i, bool negative, float &d) {
+  // Powers of ten that are exactly representable as binary32 values: 10^k is
+  // exact as long as k <= 10 (5^10 = 9765625 < 2**24 while 5^11 > 2**24).
+  static constexpr float power_of_ten_float[] = {
+      1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+      1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+  // The range of powers of ten that a non-zero, finite binary32 value can be
+  // built from. Anything smaller rounds to zero, anything larger is infinite.
+  // These are the smallest_power_of_ten()/largest_power_of_ten() constants that
+  // fast_float uses for binary32.
+  constexpr int smallest_power_binary32 = -65;
+  constexpr int largest_power_binary32 = 38;
+
+  // We start with the fast path described in
+  // Clinger WD. How to read floating point numbers accurately.
+  // ACM SIGPLAN Notices. 1990
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+  // We cannot be certain that x/y is rounded to nearest.
+  if (0 <= power && power <= 10 && i <= 16777215)
+#else
+  if (-10 <= power && power <= 10 && i <= 16777215)
+#endif
+  {
+    // Convert the integer into a float. This is lossless since
+    // 0 <= i <= 2^24 - 1.
+    d = float(i);
+    // Both d and the power of ten are exactly representable as binary32
+    // values, so the product (or quotient) is correctly rounded.
+    if (power < 0) {
+      d = d / power_of_ten_float[-power];
+    } else {
+      d = d * power_of_ten_float[power];
+    }
+    if (negative) {
+      d = -d;
+    }
+    return true;
+  }
+
+  // The fast path has failed, so we fall back on the Eisel-Lemire algorithm.
+  // It needs i > 0 (so that the leading bit of i can be normalized), so we
+  // handle i == 0 separately. We also handle the powers of ten that are so
+  // small (or so large) that the answer is zero (or infinite) whatever the
+  // mantissa is.
+  if (i == 0 || power < smallest_power_binary32) {
+    d = negative ? -0.0f : 0.0f;
+    return true;
+  }
+  if (power > largest_power_binary32) {
+    // We have, for sure, an infinite value.
+    return false;
+  }
+
+  // The exponent is 128 + 63 + power + floor(log(5**power)/log(2)).
+  // The 128 comes from the ieee32 standard (the minimal exponent is -127).
+  // The 63 comes from the fact that we use a 64-bit word.
+  // See compute_float_64 for a discussion of the magical 152170 + 65536.
+  int64_t exponent = (((152170 + 65536) * power) >> 16) + 128 + 63;
+
+  // We want the most significant bit of i to be 1. Shift if needed.
+  int lz = leading_zeroes(i);
+  i <<= lz;
+
+  // We want the most significant 64 bits of the product i * 5**power. It is
+  // safe to index the table because
+  // smallest_power <= smallest_power_binary32 <= power
+  //               <= largest_power_binary32 <= largest_power.
+  const uint32_t index = 2 * uint32_t(power - simdjson::internal::smallest_power);
+#if SIMDJSON_STATIC_REFLECTION
+  simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index]);
+#else
+  simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index]);
+#endif
+
+  // Unless the least significant 38 bits of the high (64-bit) part of the full
+  // product are all 1s, then we know that the most significant 26 bits are
+  // exact and no further work is needed. Having 26 bits is necessary because
+  // we need 24 bits for the mantissa but we have to have one rounding bit and
+  // we can waste a bit if the most significant bit of the product is zero.
+  // (For binary64, the same reasoning gives 55 bits and a 9-bit mask.)
+  if((firstproduct.high & 0x3FFFFFFFFF) == 0x3FFFFFFFFF) {
+    // The truncated multiplication was not accurate enough; use the next 64
+    // bits of the power of five to refine it. See compute_float_64 for a
+    // detailed discussion.
+#if SIMDJSON_STATIC_REFLECTION
+    simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index + 1]);
+#else
+    simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index + 1]);
+#endif
+    firstproduct.low += secondproduct.high;
+    if(secondproduct.high > firstproduct.low) { firstproduct.high++; }
+  }
+  uint64_t lower = firstproduct.low;
+  uint64_t upper = firstproduct.high;
+  // The final mantissa should be 24 bits with a leading 1.
+  // We shift it so that it occupies 25 bits with a leading 1.
+  ///////
+  uint64_t upperbit = upper >> 63;
+  uint64_t mantissa = upper >> (upperbit + 38); // 38 == 64 - 23 - 3
+  lz += int(1 ^ upperbit);
+
+  // Here we have mantissa < (1<<25).
+  int64_t real_exponent = exponent - lz;
+  if (simdjson_unlikely(real_exponent <= 0)) { // we have a subnormal?
+    // Here we have that real_exponent <= 0 so -real_exponent >= 0
+    if(-real_exponent + 1 >= 64) { // if we have more than 64 bits below the minimum exponent, you have a zero for sure.
+      d = negative ? -0.0f : 0.0f;
+      return true;
+    }
+    // next line is safe because -real_exponent + 1 < 64
+    mantissa >>= -real_exponent + 1;
+    // Thankfully, we can't have both "round-to-even" and subnormals because
+    // "round-to-even" only occurs for powers close to 0.
+    mantissa += (mantissa & 1); // round up
+    mantissa >>= 1;
+    // As in compute_float_64, rounding up may take us out of the subnormal
+    // range, so we can only decide after rounding.
+    real_exponent = (mantissa < (uint64_t(1) << 23)) ? 0 : 1;
+    d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+    return true;
+  }
+  // We have to round to even. The "to even" part is only a problem when we are
+  // right in between two floats, which we guard against. The bounds on the
+  // power of ten are those of fast_float's
+  // min_exponent_round_to_even()/max_exponent_round_to_even() for binary32:
+  // when q >= 0, (2m+1) must be divisible by 5^q with 5^q <= 2^25, so q <= 10;
+  // when q < 0, we need 2^24 x 5^{-q} < 2^{64}, so q >= -17.
+  if (simdjson_unlikely((lower <= 1) && (power >= -17) && (power <= 10) && ((mantissa & 3) == 1))) {
+    if((mantissa << (upperbit + 38)) == upper) {
+      mantissa &= ~uint64_t(1);   // flip it so that we do not round up
+    }
+  }
+
+  mantissa += mantissa & 1;
+  mantissa >>= 1;
+
+  // Here we have mantissa < (1<<24), unless there was an overflow
+  if (mantissa >= (uint64_t(1) << 24)) {
+    mantissa = (uint64_t(1) << 23);
+    real_exponent++;
+  }
+  mantissa &= ~(uint64_t(1) << 23);
+  // we have to check that real_exponent is in range, otherwise we bail out
+  if (simdjson_unlikely(real_exponent > 254)) {
+    // We have an infinite value!!! We could actually throw an error here if we could.
+    return false;
+  }
+  d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+  return true;
+}
+
+#if SIMDJSON_ENABLE_NAN_INF
+// Parses a nan or infinity. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, double& d) noexcept {
+  if (atomparsing::is_valid_inf_atom(src)) {
+    double inf = std::numeric_limits<double>::infinity();
+    d = negative ? -inf : inf;
+    return true;
+  }
+
+  if (atomparsing::is_valid_nan_atom(src)) {
+    d = std::numeric_limits<double>::quiet_NaN();
+    return true;
+  }
+
+  return false;
+}
+
+// Parses a nan or infinity as a binary32 value. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, float& d) noexcept {
+  if (atomparsing::is_valid_inf_atom(src)) {
+    float inf = std::numeric_limits<float>::infinity();
+    d = negative ? -inf : inf;
+    return true;
+  }
+
+  if (atomparsing::is_valid_nan_atom(src)) {
+    d = std::numeric_limits<float>::quiet_NaN();
+    return true;
+  }
+
+  return false;
+}
+#endif
+
 // We call a fallback floating-point parser that might be slow. Note
 // it will accept JSON numbers, but the JSON spec. is more restrictive so
 // before you call parse_float_fallback, you need to have validated the input
@@ -15538,6 +17341,16 @@ static bool parse_float_fallback(const uint8_t *ptr, const uint8_t *end_ptr, dou
   return !(*outDouble > (std::numeric_limits<double>::max)() || *outDouble < std::numeric_limits<double>::lowest());
 }

+// Same as parse_float_fallback, but for binary32 (float) values. Going through
+// the binary64 fallback and then rounding to binary32 would be subject to
+// double rounding, so we run the fallback algorithm directly on binary32.
+static bool parse_float_fallback(const uint8_t *ptr, float *outFloat) {
+  *outFloat = simdjson::internal::from_chars_float(reinterpret_cast<const char *>(ptr));
+  // We do not accept infinite values. See the binary64 version above for why we
+  // do not use std::isfinite.
+  return !(*outFloat > (std::numeric_limits<float>::max)() || *outFloat < std::numeric_limits<float>::lowest());
+}
+
 // check quickly whether the next 8 chars are made of digits
 // at a glance, it looks better than Mula's
 // http://0x80.pl/articles/swar-digits-validate.html
@@ -15556,6 +17369,28 @@ simdjson_inline bool is_made_of_eight_digits_fast(const uint8_t *chars) {
           0x3333333333333333);
 }

+// The same idea for four characters. A block of eight only bites when eight
+// digits are left, so a run of 4 to 7 digits used to fall back to one digit at a
+// time; taking four of them at once is what makes the tail of a long fraction
+// cheap. Adapted from fast_float, credit @aqrit.
+simdjson_inline bool is_made_of_four_digits_fast(const uint8_t *chars) {
+  uint32_t val;
+  // Reads up to 3 bytes beyond the digits, which SIMDJSON_PADDING covers.
+  static_assert(3 <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be bigger than 3");
+  std::memcpy(&val, chars, 4);
+  return !(((val + 0x46464646) | (val - 0x30303030)) & 0x80808080);
+}
+
+// Only call this when is_made_of_four_digits_fast() says the four characters are
+// digits. Little-endian: guarded by SIMDJSON_SWAR_NUMBER_PARSING.
+simdjson_inline uint32_t parse_four_digits_unrolled(const uint8_t *chars) {
+  uint32_t val;
+  std::memcpy(&val, chars, 4);
+  val -= 0x30303030;
+  val = (val * 10) + (val >> 8);
+  return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
 template<typename I>
 SIMDJSON_NO_SANITIZE_UNDEFINED // We deliberately allow overflow here and check later
 simdjson_inline bool parse_digit(const uint8_t c, I &i) {
@@ -15572,26 +17407,73 @@ simdjson_inline bool is_digit(const uint8_t c) {
   return static_cast<uint8_t>(c - '0') <= 9;
 }

-simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
-  // we continue with the fiction that we have an integer. If the
-  // floating point number is representable as x * 10^z for some integer
-  // z that fits in 53 bits, then we will be able to convert back the
-  // the integer into a float in a lossless manner.
-  const uint8_t *const first_after_period = p;
+// Consumes a run of digits into i, eight at a time while eight are available,
+// then four, then one at a time. Overflow is deliberate: the caller counts the
+// digits and falls back when there are too many for a 64-bit mantissa.
+//
+// Long runs of digits are what a float-heavy document is made of: the fraction
+// of a binary64 printed to full precision is 15 to 17 digits, and reading those
+// one at a time is the single largest cost in parse_double(). Integer parts are
+// left alone: they are usually a handful of digits, and there the check for a
+// block of eight is wasted work.
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_fraction_digits(const uint8_t *&p, uint64_t &i) {
+#ifdef SIMDJSON_SWAR_NUMBER_PARSING
+#if SIMDJSON_SWAR_NUMBER_PARSING
+  while (is_made_of_eight_digits_fast(p)) {
+    i = i * 100000000 + parse_eight_digits_unrolled(p);
+    p += 8;
+  }
+  // A 4 to 7 digit remainder is the common case once the blocks of eight are
+  // gone: 15 fraction digits leave 7, and 17 leave 1 after two blocks.
+  if (is_made_of_four_digits_fast(p)) {
+    i = i * 10000 + parse_four_digits_unrolled(p);
+    p += 4;
+  }
+#endif // SIMDJSON_SWAR_NUMBER_PARSING
+#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
+  while (parse_digit(*p, i)) { p++; }
+}
+
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_float_integer_digits(const uint8_t *&p, uint64_t &i,
+                                                bool &leading_zero) {
+  if (!parse_digit(*p, i)) { leading_zero = true; return; }
+  p++;
+  leading_zero = (i == 0);
+  while (parse_digit(*p, i)) { p++; }
+}

+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_integer_digits(const uint8_t *&p, uint64_t &i) {
 #ifdef SIMDJSON_SWAR_NUMBER_PARSING
 #if SIMDJSON_SWAR_NUMBER_PARSING
-  // this helps if we have lots of decimals!
-  // this turns out to be frequent enough.
+  // Identifiers, timestamps and counters often have eight digits or more.
+  // Prior related work: jsonifier parses integers as eight-digit SWAR words
+  // (str_to_i.hpp, https://github.com/nihilai-collective/Jsonifier). This
+  // takes one such word with parse_eight_digits_unrolled, the routine
+  // simdjson uses for long fractions.
   if (is_made_of_eight_digits_fast(p)) {
     i = i * 100000000 + parse_eight_digits_unrolled(p);
     p += 8;
   }
+  const uint8_t *const swar_end = p + 8;
+  while (p < swar_end && is_made_of_four_digits_fast(p)) {
+    i = i * 10000 + parse_four_digits_unrolled(p);
+    p += 4;
+  }
 #endif // SIMDJSON_SWAR_NUMBER_PARSING
 #endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
-  // Unrolling the first digit makes a small difference on some implementations (e.g. westmere)
-  if (parse_digit(*p, i)) { ++p; }
   while (parse_digit(*p, i)) { p++; }
+}
+
+simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
+  // we continue with the fiction that we have an integer. If the
+  // floating point number is representable as x * 10^z for some integer
+  // z that fits in 53 bits, then we will be able to convert back the
+  // the integer into a float in a lossless manner.
+  const uint8_t *const first_after_period = p;
+  parse_fraction_digits(p, i);
   exponent = first_after_period - p;
   // Decimal without digits (123.) is illegal
   if (exponent == 0) {
@@ -15680,7 +17562,7 @@ simdjson_inline size_t significant_digits(const uint8_t * start_digits, size_t d
 } // unnamed namespace

 /** @private */
-static error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
+inline error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
   if (parse_float_fallback(src, answer)) {
     return SUCCESS;
   }
@@ -15763,15 +17645,17 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
   return SUCCESS;              // always succeeds
 }

-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept  { return false; }
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept  { return false; }
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept { return number_type::signed_integer; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline bool is_negative(const uint8_t *) noexcept  { return false; }
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t *) noexcept  { return false; }
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t *) noexcept { return number_type::signed_integer; }
 #else

 // parse the number at src
@@ -15802,7 +17686,21 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
   size_t digit_count = size_t(p - start_digits);
-  if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) { return INVALID_NUMBER(src); }
+  if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // By this point, we know that our input does not begin with a digit. We will attempt
+    // to handle NaN/Infinity.
+
+    double d;
+    if (compute_nan_inf(p, negative, d)) {
+      writer.append_double(d);
+      return SUCCESS;
+    }
+#endif
+
+    return INVALID_NUMBER(src);
+  }

   //
   // Handle floats if there is a . or e (or both)
@@ -15851,7 +17749,7 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
     // - Therefore, if the number is positive and lower than that, it's overflow.
     // - The value we are looking at is less than or equal to INT64_MAX.
     //
-    }  else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return INVALID_NUMBER(src); }
+    }  else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return BIGINT_NUMBER(src); }
   }

   // Write unsigned if it does not fit in a signed integer.
@@ -15950,7 +17848,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const u
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -16048,7 +17946,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_stri
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -16103,7 +18001,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uin
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -16189,7 +18087,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = src;
   uint64_t i = 0;
-  while (parse_digit(*src, i)) { src++; }
+  parse_integer_digits(src, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -16229,11 +18127,20 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   //
   uint64_t i = 0;
   const uint8_t *p = src;
-  p += parse_digit(*p, i);
-  bool leading_zero = (i == 0);
-  while (parse_digit(*p, i)) { p++; }
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
   // no integer digits, or 0123 (zero must be solo)
-  if ( p == src ) { return INCORRECT_TYPE; }
+  if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no loading digits, the number may be nan or infinity.
+    // Attempt to compute those, and return on success.
+    double d;
+    if (compute_nan_inf(p, negative, d)) { return d; }
+#endif
+
+    return INCORRECT_TYPE;
+  }
   if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }

   //
@@ -16244,9 +18151,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   if (simdjson_likely(*p == '.')) {
     p++;
     const uint8_t *start_decimal_digits = p;
-    if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
-    p++;
-    while (parse_digit(*p, i)) { p++; }
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
     exponent = -(p - start_decimal_digits);

     // Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -16295,67 +18201,12 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   return d;
 }

-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
-  return (*src == '-');
-}
-
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept {
-  bool negative = (*src == '-');
-  src += uint8_t(negative);
-  const uint8_t *p = src;
-  while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
-  if ( p == src ) { return NUMBER_ERROR; }
-  if (jsoncharutils::is_structural_or_whitespace(*p)) { return true; }
-  return false;
-}
-
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept {
-  bool negative = (*src == '-');
-  src += uint8_t(negative);
-  const uint8_t *p = src;
-  while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
-  size_t digit_count = size_t(p - src);
-  if ( p == src ) { return NUMBER_ERROR; }
-  if (jsoncharutils::is_structural_or_whitespace(*p)) {
-    static const uint8_t * smaller_big_integer = reinterpret_cast<const uint8_t *>("9223372036854775808");
-    // We have an integer.
-    if(simdjson_unlikely(digit_count > 20)) {
-      return number_type::big_integer;
-    }
-    // If the number is negative and valid, it must be a signed integer.
-    if(negative) {
-      if (simdjson_unlikely(digit_count > 19)) return number_type::big_integer;
-      if (simdjson_unlikely(digit_count == 19 && memcmp(src, smaller_big_integer, 19) > 0)) {
-        return number_type::big_integer;
-      }
-#if SIMDJSON_MINUS_ZERO_AS_FLOAT
-      if(digit_count == 1 && src[0] == '0') {
-        // We have to write -0.0 instead of 0
-        return number_type::floating_point_number;
-      }
-#endif
-      return number_type::signed_integer;
-    }
-    // Let us check if we have a big integer (>=2**64).
-    static const uint8_t * two_to_sixtyfour = reinterpret_cast<const uint8_t *>("18446744073709551616");
-    if((digit_count > 20) || (digit_count == 20 && memcmp(src, two_to_sixtyfour, 20) >= 0)) {
-      return number_type::big_integer;
-    }
-    // The number is positive and smaller than 18446744073709551616 (or 2**64).
-    // We want values larger or equal to 9223372036854775808 to be unsigned
-    // integers, and the other values to be signed integers.
-    if((digit_count == 20) || (digit_count >= 19 && memcmp(src, smaller_big_integer, 19) >= 0)) {
-      return number_type::unsigned_integer;
-    }
-    return number_type::signed_integer;
-  }
-  // Hopefully, we have 'e' or 'E' or '.'.
-  return number_type::floating_point_number;
-}
-
-// Never read at src_end or beyond
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * src, const uint8_t * const src_end) noexcept {
-  if(src == src_end) { return NUMBER_ERROR; }
+// Parse a JSON number into a binary32 (float) value.
+//
+// This mirrors parse_double, but it rounds to binary32 directly instead of
+// rounding to binary64 and then to binary32: the latter is subject to double
+// rounding and would not always produce the nearest float.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * src) noexcept {
   //
   // Check for minus sign
   //
@@ -16367,91 +18218,20 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   //
   uint64_t i = 0;
   const uint8_t *p = src;
-  if(p == src_end) { return NUMBER_ERROR; }
-  p += parse_digit(*p, i);
-  bool leading_zero = (i == 0);
-  while ((p != src_end) && parse_digit(*p, i)) { p++; }
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
   // no integer digits, or 0123 (zero must be solo)
-  if ( p == src ) { return INCORRECT_TYPE; }
-  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+  if ( p == src ) {

-  //
-  // Parse the decimal part.
-  //
-  int64_t exponent = 0;
-  bool overflow;
-  if (simdjson_likely((p != src_end) && (*p == '.'))) {
-    p++;
-    const uint8_t *start_decimal_digits = p;
-    if ((p == src_end) || !parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
-    p++;
-    while ((p != src_end) && parse_digit(*p, i)) { p++; }
-    exponent = -(p - start_decimal_digits);
-
-    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
-    overflow = p-src-1 > 19;
-    if (simdjson_unlikely(overflow && leading_zero)) {
-      // Skip leading 0.00000 and see if it still overflows
-      const uint8_t *start_digits = src + 2;
-      while (*start_digits == '0') { start_digits++; }
-      overflow = start_digits-src > 19;
-    }
-  } else {
-    overflow = p-src > 19;
-  }
-
-  //
-  // Parse the exponent
-  //
-  if ((p != src_end) && (*p == 'e' || *p == 'E')) {
-    p++;
-    if(p == src_end) { return NUMBER_ERROR; }
-    bool exp_neg = *p == '-';
-    p += exp_neg || *p == '+';
-
-    uint64_t exp = 0;
-    const uint8_t *start_exp_digits = p;
-    while ((p != src_end) && parse_digit(*p, exp)) { p++; }
-    // no exp digits, or 20+ exp digits
-    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
-
-    exponent += exp_neg ? 0-exp : exp;
-  }
-
-  if ((p != src_end) && jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
-
-  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, the number may be nan or infinity.
+    // Attempt to compute those, and return on success.
+    float f;
+    if (compute_nan_inf(p, negative, f)) { return f; }
+#endif

-  //
-  // Assemble (or slow-parse) the float
-  //
-  double d;
-  if (simdjson_likely(!overflow)) {
-    if (compute_float_64(exponent, i, negative, d)) { return d; }
-  }
-  if (!parse_float_fallback(src - uint8_t(negative), src_end, &d)) {
-    return NUMBER_ERROR;
+    return INCORRECT_TYPE;
   }
-  return d;
-}
-
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * src) noexcept {
-  //
-  // Check for minus sign
-  //
-  bool negative = (*(src + 1) == '-');
-  src += uint8_t(negative) + 1;
-
-  //
-  // Parse the integer part.
-  //
-  uint64_t i = 0;
-  const uint8_t *p = src;
-  p += parse_digit(*p, i);
-  bool leading_zero = (i == 0);
-  while (parse_digit(*p, i)) { p++; }
-  // no integer digits, or 0123 (zero must be solo)
-  if ( p == src ) { return INCORRECT_TYPE; }
   if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }

   //
@@ -16462,9 +18242,239 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   if (simdjson_likely(*p == '.')) {
     p++;
     const uint8_t *start_decimal_digits = p;
-    if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
-    p++;
-    while (parse_digit(*p, i)) { p++; }
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = p-start_digits > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if (*p == 'e' || *p == 'E') {
+    p++;
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while (parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  float d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_32(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
+simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
+  return (*src == '-');
+}
+
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept {
+  bool negative = (*src == '-');
+  src += uint8_t(negative);
+  const uint8_t *p = src;
+  while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
+  if ( p == src ) { return NUMBER_ERROR; }
+  if (jsoncharutils::is_structural_or_whitespace(*p)) { return true; }
+  return false;
+}
+
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept {
+  bool negative = (*src == '-');
+  src += uint8_t(negative);
+  const uint8_t *p = src;
+  while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
+  size_t digit_count = size_t(p - src);
+  if ( p == src ) { return NUMBER_ERROR; }
+  if (jsoncharutils::is_structural_or_whitespace(*p)) {
+    static const uint8_t * smaller_big_integer = reinterpret_cast<const uint8_t *>("9223372036854775808");
+    // We have an integer.
+    if(simdjson_unlikely(digit_count > 20)) {
+      return number_type::big_integer;
+    }
+    // If the number is negative and valid, it must be a signed integer.
+    if(negative) {
+      if (simdjson_unlikely(digit_count > 19)) return number_type::big_integer;
+      if (simdjson_unlikely(digit_count == 19 && memcmp(src, smaller_big_integer, 19) > 0)) {
+        return number_type::big_integer;
+      }
+#if SIMDJSON_MINUS_ZERO_AS_FLOAT
+      if(digit_count == 1 && src[0] == '0') {
+        // We have to write -0.0 instead of 0
+        return number_type::floating_point_number;
+      }
+#endif
+      return number_type::signed_integer;
+    }
+    // Let us check if we have a big integer (>=2**64).
+    static const uint8_t * two_to_sixtyfour = reinterpret_cast<const uint8_t *>("18446744073709551616");
+    if((digit_count > 20) || (digit_count == 20 && memcmp(src, two_to_sixtyfour, 20) >= 0)) {
+      return number_type::big_integer;
+    }
+    // The number is positive and smaller than 18446744073709551616 (or 2**64).
+    // We want values larger or equal to 9223372036854775808 to be unsigned
+    // integers, and the other values to be signed integers.
+    if((digit_count == 20) || (digit_count >= 19 && memcmp(src, smaller_big_integer, 19) >= 0)) {
+      return number_type::unsigned_integer;
+    }
+    return number_type::signed_integer;
+  }
+  // Hopefully, we have 'e' or 'E' or '.'.
+  return number_type::floating_point_number;
+}
+
+// Never read at src_end or beyond
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * src, const uint8_t * const src_end) noexcept {
+  if(src == src_end) { return NUMBER_ERROR; }
+  //
+  // Check for minus sign
+  //
+  bool negative = (*src == '-');
+  src += uint8_t(negative);
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  if(p == src_end) { return NUMBER_ERROR; }
+  p += parse_digit(*p, i);
+  bool leading_zero = (i == 0);
+  while ((p != src_end) && parse_digit(*p, i)) { p++; }
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) { return INCORRECT_TYPE; }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely((p != src_end) && (*p == '.'))) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    if ((p == src_end) || !parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
+    p++;
+    while ((p != src_end) && parse_digit(*p, i)) { p++; }
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = start_digits-src > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if ((p != src_end) && (*p == 'e' || *p == 'E')) {
+    p++;
+    if(p == src_end) { return NUMBER_ERROR; }
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while ((p != src_end) && parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if ((p != src_end) && jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  double d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_64(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), src_end, &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * src) noexcept {
+  //
+  // Check for minus sign
+  //
+  bool negative = (*(src + 1) == '-');
+  src += uint8_t(negative) + 1;
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, attempt to parse numbers that are either
+    // NaN or Infinity
+    if (atomparsing::is_valid_inf_in_string(src)) {
+      double inf = std::numeric_limits<double>::infinity();
+      return negative ? -inf : inf;
+    }
+
+    if (atomparsing::is_valid_nan_in_string(src)) {
+      return std::numeric_limits<double>::quiet_NaN();
+    }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
     exponent = -(p - start_decimal_digits);

     // Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -16513,6 +18523,99 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   return d;
 }

+// Parse a JSON number held inside a JSON string into a binary32 (float) value.
+// See parse_float for why we do not simply round parse_double_in_string.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * src) noexcept {
+  //
+  // Check for minus sign
+  //
+  bool negative = (*(src + 1) == '-');
+  src += uint8_t(negative) + 1;
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, attempt to parse numbers that are either
+    // NaN or Infinity
+    if (atomparsing::is_valid_inf_in_string(src)) {
+      float inf = std::numeric_limits<float>::infinity();
+      return negative ? -inf : inf;
+    }
+
+    if (atomparsing::is_valid_nan_in_string(src)) {
+      return std::numeric_limits<float>::quiet_NaN();
+    }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = p-start_digits > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if (*p == 'e' || *p == 'E') {
+    p++;
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while (parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if (*p != '"') { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  float d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_32(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
 } // unnamed namespace
 #endif // SIMDJSON_SKIPNUMBERPARSING

@@ -17109,6 +19212,9 @@ namespace atomparsing {
 // to the compile-time constant 1936482662.
 simdjson_inline uint32_t string_to_uint32(const char* str) { uint32_t val; std::memcpy(&val, str, sizeof(uint32_t)); return val; }

+// Acts on the same principle as string_to_uint32, but on an 8-byte block of memory
+simdjson_inline uint64_t string_to_uint64(const char* str) { uint64_t val; std::memcpy(&val, str, sizeof(uint64_t)); return val; }
+

 // Again in str4ncmp we use a memcpy to avoid undefined behavior. The memcpy may appear expensive.
 // Yet all decent optimizing compilers will compile memcpy to a single instruction, just about.
@@ -17120,6 +19226,28 @@ simdjson_inline uint32_t str4ncmp(const uint8_t *src, const char* atom) {
   return srcval ^ string_to_uint32(atom);
 }

+// Checks that the first 8 characters of the input string match the given atom in a case-insensitive manner.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint64_t str8ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+  uint64_t srcval; // we want to avoid unaligned 32-bit loads (undefined in C/C++)
+  static_assert(sizeof(uint64_t) <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be larger than 8 bytes");
+  std::memcpy(&srcval, src, sizeof(uint64_t));
+
+  return (srcval | 0x2020202020202020ull) ^ string_to_uint64(atom);
+}
+
+// Checks that the first 3 characters of 'src' match 'atom' in a case-insensitive way.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint32_t str3ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+  return ((src[0] | 0x20) ^ atom[0]) //
+       | ((src[1] | 0x20) ^ atom[1]) //
+       | ((src[2] | 0x20) ^ atom[2]);
+}
+
 simdjson_warn_unused
 simdjson_inline bool is_valid_true_atom(const uint8_t *src) {
   return (str4ncmp(src, "true") | jsoncharutils::is_not_structural_or_whitespace(src[4])) == 0;
@@ -17156,6 +19284,71 @@ simdjson_inline bool is_valid_null_atom(const uint8_t *src, size_t len) {
   else { return false; }
 }

+#if SIMDJSON_ENABLE_NAN_INF
+// "nan" is 3 bytes; we check characters and then verify the next
+// character is structural or whitespace. We accept both "nan" and "NaN".
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src) {
+  return (str3ncmp_case_insensitive(src, "nan")
+        | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+}
+
+// checks that the next four characters of a string are 'nan"', where the 'nan'
+// is checked in a case-insensitive way.
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_in_string(const uint8_t *src) {
+  return (str3ncmp_case_insensitive(src, "nan") | (src[3] ^ '"')) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src, size_t len) {
+  if (len > 3) { return is_valid_nan_atom(src); }
+  if (len == 3) { return str3ncmp_case_insensitive(src, "nan") == 0; }
+  return false;
+}
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src) {
+  bool is_short_inf = (str3ncmp_case_insensitive(src, "inf")
+                     | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+  if(is_short_inf) return true;
+
+  // Check for 'infinity' (any capitalization)
+  return (str8ncmp_case_insensitive(src, "infinity") | jsoncharutils::is_not_structural_or_whitespace(src[8])) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_in_string(const uint8_t *src) {
+  bool is_short_inf = (str3ncmp_case_insensitive(src, "inf") | (src[3] ^ '"')) == 0;
+  if(is_short_inf) return true;
+
+  return (str8ncmp_case_insensitive(src, "infinity") | (src[8] ^ '"')) == 0;
+}
+
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src, size_t len) {
+  if (len > 8) { return is_valid_inf_atom(src); }
+  if (len == 8 && str8ncmp_case_insensitive(src, "infinity") == 0) {
+    return true;
+  }
+  if (len > 3) {
+    return (str3ncmp_case_insensitive(src, "inf")
+          | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+  }
+  if (len == 3) { return str3ncmp_case_insensitive(src, "inf") == 0; }
+  return false;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
 } // namespace atomparsing
 } // unnamed namespace
 } // namespace fallback
@@ -17246,7 +19439,7 @@ inline simdjson_warn_unused error_code dom_parser_implementation::set_capacity(s
 }

 inline simdjson_warn_unused error_code dom_parser_implementation::set_max_depth(size_t max_depth) noexcept {
-  if(max_depth == 0 || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
+  if(max_depth == 0 || max_depth > SIMDJSON_MAX_DEPTH || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
   // Stage 2 stacks
   open_containers.reset(new (std::nothrow) open_container[max_depth]);
   is_array.reset(new (std::nothrow) bool[max_depth]);
@@ -17425,6 +19618,7 @@ protected:
 /* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_NUMBERPARSING_H */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/base.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/jsoncharutils.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/atomparsing.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/internal/numberparsing_tables.h" */
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */

@@ -17464,6 +19658,18 @@ simdjson_inline double to_double(uint64_t mantissa, uint64_t real_exponent, bool
     return d;
 }

+// Convert a mantissa, an exponent and a sign bit into an ieee32 float (binary32).
+// The real_exponent needs to be in [0, 254] (technically real_exponent = 255 would be acceptable).
+// The mantissa should be in [0,1<<24). The bit at index (1U << 23) will be zeroed.
+simdjson_inline float to_float(uint32_t mantissa, uint32_t real_exponent, bool negative) {
+    float f;
+    mantissa &= ~(uint32_t(1) << 23);
+    mantissa |= real_exponent << 23;
+    mantissa |= ((static_cast<uint32_t>(negative)) << 31);
+    std::memcpy(&f, &mantissa, sizeof(f));
+    return f;
+}
+
 // Attempts to compute i * 10^(power) exactly; and if "negative" is
 // true, negate the result.
 // This function will only work in some cases, when it does not work, success is
@@ -17720,6 +19926,219 @@ simdjson_inline bool compute_float_64(int64_t power, uint64_t i, bool negative,
   return true;
 }

+// Attempts to compute i * 10^(power) as a binary32 (float) value; and if
+// "negative" is true, negate the result.
+//
+// This is the same Eisel-Lemire algorithm as compute_float_64, retargeted at
+// binary32. It is adapted from fast_float
+// (https://github.com/fastfloat/fast_float), where the routine is written once
+// and instantiated for each binary format. We only need 24 bits of mantissa
+// instead of 53, so we ask the truncated product for 24 + 2 = 26 accurate bits
+// (plus one bit that may be lost to the "upperbit" shift), and we shift the
+// product right by (upperbit + 64 - 23 - 3) instead of (upperbit + 64 - 52 - 3).
+//
+// The power-of-five table is shared with the binary64 code: the useful range
+// for binary32, [smallest_power_binary32, largest_power_binary32], is a strict
+// subset of [smallest_power, largest_power].
+//
+// The function returns false when the result would be infinite: simdjson
+// refuses to parse infinite values, so the caller reports an error. Unlike
+// compute_float_64, a false return never means "try harder": the accuracy of
+// the two-limb product is guaranteed by Noble Mushtak and Daniel Lemire,
+// "Fast Number Parsing Without Fallback" (https://arxiv.org/abs/2212.06644).
+// The caller must still fall back to a slow path when the 64-bit mantissa i
+// was truncated (more than 19 significant digits).
+//
+// We assume that power is in the [smallest_power, largest_power] interval: the
+// caller is responsible for this check.
+simdjson_inline bool compute_float_32(int64_t power, uint64_t i, bool negative, float &d) {
+  // Powers of ten that are exactly representable as binary32 values: 10^k is
+  // exact as long as k <= 10 (5^10 = 9765625 < 2**24 while 5^11 > 2**24).
+  static constexpr float power_of_ten_float[] = {
+      1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+      1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+  // The range of powers of ten that a non-zero, finite binary32 value can be
+  // built from. Anything smaller rounds to zero, anything larger is infinite.
+  // These are the smallest_power_of_ten()/largest_power_of_ten() constants that
+  // fast_float uses for binary32.
+  constexpr int smallest_power_binary32 = -65;
+  constexpr int largest_power_binary32 = 38;
+
+  // We start with the fast path described in
+  // Clinger WD. How to read floating point numbers accurately.
+  // ACM SIGPLAN Notices. 1990
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+  // We cannot be certain that x/y is rounded to nearest.
+  if (0 <= power && power <= 10 && i <= 16777215)
+#else
+  if (-10 <= power && power <= 10 && i <= 16777215)
+#endif
+  {
+    // Convert the integer into a float. This is lossless since
+    // 0 <= i <= 2^24 - 1.
+    d = float(i);
+    // Both d and the power of ten are exactly representable as binary32
+    // values, so the product (or quotient) is correctly rounded.
+    if (power < 0) {
+      d = d / power_of_ten_float[-power];
+    } else {
+      d = d * power_of_ten_float[power];
+    }
+    if (negative) {
+      d = -d;
+    }
+    return true;
+  }
+
+  // The fast path has failed, so we fall back on the Eisel-Lemire algorithm.
+  // It needs i > 0 (so that the leading bit of i can be normalized), so we
+  // handle i == 0 separately. We also handle the powers of ten that are so
+  // small (or so large) that the answer is zero (or infinite) whatever the
+  // mantissa is.
+  if (i == 0 || power < smallest_power_binary32) {
+    d = negative ? -0.0f : 0.0f;
+    return true;
+  }
+  if (power > largest_power_binary32) {
+    // We have, for sure, an infinite value.
+    return false;
+  }
+
+  // The exponent is 128 + 63 + power + floor(log(5**power)/log(2)).
+  // The 128 comes from the ieee32 standard (the minimal exponent is -127).
+  // The 63 comes from the fact that we use a 64-bit word.
+  // See compute_float_64 for a discussion of the magical 152170 + 65536.
+  int64_t exponent = (((152170 + 65536) * power) >> 16) + 128 + 63;
+
+  // We want the most significant bit of i to be 1. Shift if needed.
+  int lz = leading_zeroes(i);
+  i <<= lz;
+
+  // We want the most significant 64 bits of the product i * 5**power. It is
+  // safe to index the table because
+  // smallest_power <= smallest_power_binary32 <= power
+  //               <= largest_power_binary32 <= largest_power.
+  const uint32_t index = 2 * uint32_t(power - simdjson::internal::smallest_power);
+#if SIMDJSON_STATIC_REFLECTION
+  simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index]);
+#else
+  simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index]);
+#endif
+
+  // Unless the least significant 38 bits of the high (64-bit) part of the full
+  // product are all 1s, then we know that the most significant 26 bits are
+  // exact and no further work is needed. Having 26 bits is necessary because
+  // we need 24 bits for the mantissa but we have to have one rounding bit and
+  // we can waste a bit if the most significant bit of the product is zero.
+  // (For binary64, the same reasoning gives 55 bits and a 9-bit mask.)
+  if((firstproduct.high & 0x3FFFFFFFFF) == 0x3FFFFFFFFF) {
+    // The truncated multiplication was not accurate enough; use the next 64
+    // bits of the power of five to refine it. See compute_float_64 for a
+    // detailed discussion.
+#if SIMDJSON_STATIC_REFLECTION
+    simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index + 1]);
+#else
+    simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index + 1]);
+#endif
+    firstproduct.low += secondproduct.high;
+    if(secondproduct.high > firstproduct.low) { firstproduct.high++; }
+  }
+  uint64_t lower = firstproduct.low;
+  uint64_t upper = firstproduct.high;
+  // The final mantissa should be 24 bits with a leading 1.
+  // We shift it so that it occupies 25 bits with a leading 1.
+  ///////
+  uint64_t upperbit = upper >> 63;
+  uint64_t mantissa = upper >> (upperbit + 38); // 38 == 64 - 23 - 3
+  lz += int(1 ^ upperbit);
+
+  // Here we have mantissa < (1<<25).
+  int64_t real_exponent = exponent - lz;
+  if (simdjson_unlikely(real_exponent <= 0)) { // we have a subnormal?
+    // Here we have that real_exponent <= 0 so -real_exponent >= 0
+    if(-real_exponent + 1 >= 64) { // if we have more than 64 bits below the minimum exponent, you have a zero for sure.
+      d = negative ? -0.0f : 0.0f;
+      return true;
+    }
+    // next line is safe because -real_exponent + 1 < 64
+    mantissa >>= -real_exponent + 1;
+    // Thankfully, we can't have both "round-to-even" and subnormals because
+    // "round-to-even" only occurs for powers close to 0.
+    mantissa += (mantissa & 1); // round up
+    mantissa >>= 1;
+    // As in compute_float_64, rounding up may take us out of the subnormal
+    // range, so we can only decide after rounding.
+    real_exponent = (mantissa < (uint64_t(1) << 23)) ? 0 : 1;
+    d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+    return true;
+  }
+  // We have to round to even. The "to even" part is only a problem when we are
+  // right in between two floats, which we guard against. The bounds on the
+  // power of ten are those of fast_float's
+  // min_exponent_round_to_even()/max_exponent_round_to_even() for binary32:
+  // when q >= 0, (2m+1) must be divisible by 5^q with 5^q <= 2^25, so q <= 10;
+  // when q < 0, we need 2^24 x 5^{-q} < 2^{64}, so q >= -17.
+  if (simdjson_unlikely((lower <= 1) && (power >= -17) && (power <= 10) && ((mantissa & 3) == 1))) {
+    if((mantissa << (upperbit + 38)) == upper) {
+      mantissa &= ~uint64_t(1);   // flip it so that we do not round up
+    }
+  }
+
+  mantissa += mantissa & 1;
+  mantissa >>= 1;
+
+  // Here we have mantissa < (1<<24), unless there was an overflow
+  if (mantissa >= (uint64_t(1) << 24)) {
+    mantissa = (uint64_t(1) << 23);
+    real_exponent++;
+  }
+  mantissa &= ~(uint64_t(1) << 23);
+  // we have to check that real_exponent is in range, otherwise we bail out
+  if (simdjson_unlikely(real_exponent > 254)) {
+    // We have an infinite value!!! We could actually throw an error here if we could.
+    return false;
+  }
+  d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+  return true;
+}
+
+#if SIMDJSON_ENABLE_NAN_INF
+// Parses a nan or infinity. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, double& d) noexcept {
+  if (atomparsing::is_valid_inf_atom(src)) {
+    double inf = std::numeric_limits<double>::infinity();
+    d = negative ? -inf : inf;
+    return true;
+  }
+
+  if (atomparsing::is_valid_nan_atom(src)) {
+    d = std::numeric_limits<double>::quiet_NaN();
+    return true;
+  }
+
+  return false;
+}
+
+// Parses a nan or infinity as a binary32 value. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, float& d) noexcept {
+  if (atomparsing::is_valid_inf_atom(src)) {
+    float inf = std::numeric_limits<float>::infinity();
+    d = negative ? -inf : inf;
+    return true;
+  }
+
+  if (atomparsing::is_valid_nan_atom(src)) {
+    d = std::numeric_limits<float>::quiet_NaN();
+    return true;
+  }
+
+  return false;
+}
+#endif
+
 // We call a fallback floating-point parser that might be slow. Note
 // it will accept JSON numbers, but the JSON spec. is more restrictive so
 // before you call parse_float_fallback, you need to have validated the input
@@ -17757,6 +20176,16 @@ static bool parse_float_fallback(const uint8_t *ptr, const uint8_t *end_ptr, dou
   return !(*outDouble > (std::numeric_limits<double>::max)() || *outDouble < std::numeric_limits<double>::lowest());
 }

+// Same as parse_float_fallback, but for binary32 (float) values. Going through
+// the binary64 fallback and then rounding to binary32 would be subject to
+// double rounding, so we run the fallback algorithm directly on binary32.
+static bool parse_float_fallback(const uint8_t *ptr, float *outFloat) {
+  *outFloat = simdjson::internal::from_chars_float(reinterpret_cast<const char *>(ptr));
+  // We do not accept infinite values. See the binary64 version above for why we
+  // do not use std::isfinite.
+  return !(*outFloat > (std::numeric_limits<float>::max)() || *outFloat < std::numeric_limits<float>::lowest());
+}
+
 // check quickly whether the next 8 chars are made of digits
 // at a glance, it looks better than Mula's
 // http://0x80.pl/articles/swar-digits-validate.html
@@ -17775,6 +20204,28 @@ simdjson_inline bool is_made_of_eight_digits_fast(const uint8_t *chars) {
           0x3333333333333333);
 }

+// The same idea for four characters. A block of eight only bites when eight
+// digits are left, so a run of 4 to 7 digits used to fall back to one digit at a
+// time; taking four of them at once is what makes the tail of a long fraction
+// cheap. Adapted from fast_float, credit @aqrit.
+simdjson_inline bool is_made_of_four_digits_fast(const uint8_t *chars) {
+  uint32_t val;
+  // Reads up to 3 bytes beyond the digits, which SIMDJSON_PADDING covers.
+  static_assert(3 <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be bigger than 3");
+  std::memcpy(&val, chars, 4);
+  return !(((val + 0x46464646) | (val - 0x30303030)) & 0x80808080);
+}
+
+// Only call this when is_made_of_four_digits_fast() says the four characters are
+// digits. Little-endian: guarded by SIMDJSON_SWAR_NUMBER_PARSING.
+simdjson_inline uint32_t parse_four_digits_unrolled(const uint8_t *chars) {
+  uint32_t val;
+  std::memcpy(&val, chars, 4);
+  val -= 0x30303030;
+  val = (val * 10) + (val >> 8);
+  return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
 template<typename I>
 SIMDJSON_NO_SANITIZE_UNDEFINED // We deliberately allow overflow here and check later
 simdjson_inline bool parse_digit(const uint8_t c, I &i) {
@@ -17791,26 +20242,73 @@ simdjson_inline bool is_digit(const uint8_t c) {
   return static_cast<uint8_t>(c - '0') <= 9;
 }

-simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
-  // we continue with the fiction that we have an integer. If the
-  // floating point number is representable as x * 10^z for some integer
-  // z that fits in 53 bits, then we will be able to convert back the
-  // the integer into a float in a lossless manner.
-  const uint8_t *const first_after_period = p;
+// Consumes a run of digits into i, eight at a time while eight are available,
+// then four, then one at a time. Overflow is deliberate: the caller counts the
+// digits and falls back when there are too many for a 64-bit mantissa.
+//
+// Long runs of digits are what a float-heavy document is made of: the fraction
+// of a binary64 printed to full precision is 15 to 17 digits, and reading those
+// one at a time is the single largest cost in parse_double(). Integer parts are
+// left alone: they are usually a handful of digits, and there the check for a
+// block of eight is wasted work.
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_fraction_digits(const uint8_t *&p, uint64_t &i) {
+#ifdef SIMDJSON_SWAR_NUMBER_PARSING
+#if SIMDJSON_SWAR_NUMBER_PARSING
+  while (is_made_of_eight_digits_fast(p)) {
+    i = i * 100000000 + parse_eight_digits_unrolled(p);
+    p += 8;
+  }
+  // A 4 to 7 digit remainder is the common case once the blocks of eight are
+  // gone: 15 fraction digits leave 7, and 17 leave 1 after two blocks.
+  if (is_made_of_four_digits_fast(p)) {
+    i = i * 10000 + parse_four_digits_unrolled(p);
+    p += 4;
+  }
+#endif // SIMDJSON_SWAR_NUMBER_PARSING
+#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
+  while (parse_digit(*p, i)) { p++; }
+}
+
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_float_integer_digits(const uint8_t *&p, uint64_t &i,
+                                                bool &leading_zero) {
+  if (!parse_digit(*p, i)) { leading_zero = true; return; }
+  p++;
+  leading_zero = (i == 0);
+  while (parse_digit(*p, i)) { p++; }
+}

+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_integer_digits(const uint8_t *&p, uint64_t &i) {
 #ifdef SIMDJSON_SWAR_NUMBER_PARSING
 #if SIMDJSON_SWAR_NUMBER_PARSING
-  // this helps if we have lots of decimals!
-  // this turns out to be frequent enough.
+  // Identifiers, timestamps and counters often have eight digits or more.
+  // Prior related work: jsonifier parses integers as eight-digit SWAR words
+  // (str_to_i.hpp, https://github.com/nihilai-collective/Jsonifier). This
+  // takes one such word with parse_eight_digits_unrolled, the routine
+  // simdjson uses for long fractions.
   if (is_made_of_eight_digits_fast(p)) {
     i = i * 100000000 + parse_eight_digits_unrolled(p);
     p += 8;
   }
+  const uint8_t *const swar_end = p + 8;
+  while (p < swar_end && is_made_of_four_digits_fast(p)) {
+    i = i * 10000 + parse_four_digits_unrolled(p);
+    p += 4;
+  }
 #endif // SIMDJSON_SWAR_NUMBER_PARSING
 #endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
-  // Unrolling the first digit makes a small difference on some implementations (e.g. westmere)
-  if (parse_digit(*p, i)) { ++p; }
   while (parse_digit(*p, i)) { p++; }
+}
+
+simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
+  // we continue with the fiction that we have an integer. If the
+  // floating point number is representable as x * 10^z for some integer
+  // z that fits in 53 bits, then we will be able to convert back the
+  // the integer into a float in a lossless manner.
+  const uint8_t *const first_after_period = p;
+  parse_fraction_digits(p, i);
   exponent = first_after_period - p;
   // Decimal without digits (123.) is illegal
   if (exponent == 0) {
@@ -17899,7 +20397,7 @@ simdjson_inline size_t significant_digits(const uint8_t * start_digits, size_t d
 } // unnamed namespace

 /** @private */
-static error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
+inline error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
   if (parse_float_fallback(src, answer)) {
     return SUCCESS;
   }
@@ -17982,15 +20480,17 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
   return SUCCESS;              // always succeeds
 }

-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept  { return false; }
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept  { return false; }
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept { return number_type::signed_integer; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline bool is_negative(const uint8_t *) noexcept  { return false; }
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t *) noexcept  { return false; }
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t *) noexcept { return number_type::signed_integer; }
 #else

 // parse the number at src
@@ -18021,7 +20521,21 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
   size_t digit_count = size_t(p - start_digits);
-  if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) { return INVALID_NUMBER(src); }
+  if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // By this point, we know that our input does not begin with a digit. We will attempt
+    // to handle NaN/Infinity.
+
+    double d;
+    if (compute_nan_inf(p, negative, d)) {
+      writer.append_double(d);
+      return SUCCESS;
+    }
+#endif
+
+    return INVALID_NUMBER(src);
+  }

   //
   // Handle floats if there is a . or e (or both)
@@ -18070,7 +20584,7 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
     // - Therefore, if the number is positive and lower than that, it's overflow.
     // - The value we are looking at is less than or equal to INT64_MAX.
     //
-    }  else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return INVALID_NUMBER(src); }
+    }  else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return BIGINT_NUMBER(src); }
   }

   // Write unsigned if it does not fit in a signed integer.
@@ -18169,7 +20683,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const u
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -18267,7 +20781,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_stri
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -18322,7 +20836,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uin
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -18408,7 +20922,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = src;
   uint64_t i = 0;
-  while (parse_digit(*src, i)) { src++; }
+  parse_integer_digits(src, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -18448,11 +20962,20 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   //
   uint64_t i = 0;
   const uint8_t *p = src;
-  p += parse_digit(*p, i);
-  bool leading_zero = (i == 0);
-  while (parse_digit(*p, i)) { p++; }
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
   // no integer digits, or 0123 (zero must be solo)
-  if ( p == src ) { return INCORRECT_TYPE; }
+  if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no loading digits, the number may be nan or infinity.
+    // Attempt to compute those, and return on success.
+    double d;
+    if (compute_nan_inf(p, negative, d)) { return d; }
+#endif
+
+    return INCORRECT_TYPE;
+  }
   if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }

   //
@@ -18463,9 +20986,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   if (simdjson_likely(*p == '.')) {
     p++;
     const uint8_t *start_decimal_digits = p;
-    if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
-    p++;
-    while (parse_digit(*p, i)) { p++; }
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
     exponent = -(p - start_decimal_digits);

     // Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -18514,6 +21036,97 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   return d;
 }

+// Parse a JSON number into a binary32 (float) value.
+//
+// This mirrors parse_double, but it rounds to binary32 directly instead of
+// rounding to binary64 and then to binary32: the latter is subject to double
+// rounding and would not always produce the nearest float.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * src) noexcept {
+  //
+  // Check for minus sign
+  //
+  bool negative = (*src == '-');
+  src += uint8_t(negative);
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, the number may be nan or infinity.
+    // Attempt to compute those, and return on success.
+    float f;
+    if (compute_nan_inf(p, negative, f)) { return f; }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = p-start_digits > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if (*p == 'e' || *p == 'E') {
+    p++;
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while (parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  float d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_32(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
 simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
   return (*src == '-');
 }
@@ -18666,11 +21279,25 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   //
   uint64_t i = 0;
   const uint8_t *p = src;
-  p += parse_digit(*p, i);
-  bool leading_zero = (i == 0);
-  while (parse_digit(*p, i)) { p++; }
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
   // no integer digits, or 0123 (zero must be solo)
-  if ( p == src ) { return INCORRECT_TYPE; }
+  if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, attempt to parse numbers that are either
+    // NaN or Infinity
+    if (atomparsing::is_valid_inf_in_string(src)) {
+      double inf = std::numeric_limits<double>::infinity();
+      return negative ? -inf : inf;
+    }
+
+    if (atomparsing::is_valid_nan_in_string(src)) {
+      return std::numeric_limits<double>::quiet_NaN();
+    }
+#endif
+
+    return INCORRECT_TYPE;
+  }
   if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }

   //
@@ -18681,9 +21308,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   if (simdjson_likely(*p == '.')) {
     p++;
     const uint8_t *start_decimal_digits = p;
-    if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
-    p++;
-    while (parse_digit(*p, i)) { p++; }
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
     exponent = -(p - start_decimal_digits);

     // Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -18732,6 +21358,99 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   return d;
 }

+// Parse a JSON number held inside a JSON string into a binary32 (float) value.
+// See parse_float for why we do not simply round parse_double_in_string.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * src) noexcept {
+  //
+  // Check for minus sign
+  //
+  bool negative = (*(src + 1) == '-');
+  src += uint8_t(negative) + 1;
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, attempt to parse numbers that are either
+    // NaN or Infinity
+    if (atomparsing::is_valid_inf_in_string(src)) {
+      float inf = std::numeric_limits<float>::infinity();
+      return negative ? -inf : inf;
+    }
+
+    if (atomparsing::is_valid_nan_in_string(src)) {
+      return std::numeric_limits<float>::quiet_NaN();
+    }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = p-start_digits > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if (*p == 'e' || *p == 'E') {
+    p++;
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while (parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if (*p != '"') { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  float d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_32(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
 } // unnamed namespace
 #endif // SIMDJSON_SKIPNUMBERPARSING

@@ -19053,16 +21772,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
 }
 #endif

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
-                                uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
-  return _addcarry_u64(0, value1, value2,
-                       reinterpret_cast<unsigned __int64 *>(result));
-#else
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-#endif
-}

 } // unnamed namespace
 } // namespace haswell
@@ -19815,6 +22524,9 @@ namespace atomparsing {
 // to the compile-time constant 1936482662.
 simdjson_inline uint32_t string_to_uint32(const char* str) { uint32_t val; std::memcpy(&val, str, sizeof(uint32_t)); return val; }

+// Acts on the same principle as string_to_uint32, but on an 8-byte block of memory
+simdjson_inline uint64_t string_to_uint64(const char* str) { uint64_t val; std::memcpy(&val, str, sizeof(uint64_t)); return val; }
+

 // Again in str4ncmp we use a memcpy to avoid undefined behavior. The memcpy may appear expensive.
 // Yet all decent optimizing compilers will compile memcpy to a single instruction, just about.
@@ -19826,6 +22538,28 @@ simdjson_inline uint32_t str4ncmp(const uint8_t *src, const char* atom) {
   return srcval ^ string_to_uint32(atom);
 }

+// Checks that the first 8 characters of the input string match the given atom in a case-insensitive manner.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint64_t str8ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+  uint64_t srcval; // we want to avoid unaligned 32-bit loads (undefined in C/C++)
+  static_assert(sizeof(uint64_t) <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be larger than 8 bytes");
+  std::memcpy(&srcval, src, sizeof(uint64_t));
+
+  return (srcval | 0x2020202020202020ull) ^ string_to_uint64(atom);
+}
+
+// Checks that the first 3 characters of 'src' match 'atom' in a case-insensitive way.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint32_t str3ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+  return ((src[0] | 0x20) ^ atom[0]) //
+       | ((src[1] | 0x20) ^ atom[1]) //
+       | ((src[2] | 0x20) ^ atom[2]);
+}
+
 simdjson_warn_unused
 simdjson_inline bool is_valid_true_atom(const uint8_t *src) {
   return (str4ncmp(src, "true") | jsoncharutils::is_not_structural_or_whitespace(src[4])) == 0;
@@ -19862,6 +22596,71 @@ simdjson_inline bool is_valid_null_atom(const uint8_t *src, size_t len) {
   else { return false; }
 }

+#if SIMDJSON_ENABLE_NAN_INF
+// "nan" is 3 bytes; we check characters and then verify the next
+// character is structural or whitespace. We accept both "nan" and "NaN".
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src) {
+  return (str3ncmp_case_insensitive(src, "nan")
+        | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+}
+
+// checks that the next four characters of a string are 'nan"', where the 'nan'
+// is checked in a case-insensitive way.
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_in_string(const uint8_t *src) {
+  return (str3ncmp_case_insensitive(src, "nan") | (src[3] ^ '"')) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src, size_t len) {
+  if (len > 3) { return is_valid_nan_atom(src); }
+  if (len == 3) { return str3ncmp_case_insensitive(src, "nan") == 0; }
+  return false;
+}
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src) {
+  bool is_short_inf = (str3ncmp_case_insensitive(src, "inf")
+                     | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+  if(is_short_inf) return true;
+
+  // Check for 'infinity' (any capitalization)
+  return (str8ncmp_case_insensitive(src, "infinity") | jsoncharutils::is_not_structural_or_whitespace(src[8])) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_in_string(const uint8_t *src) {
+  bool is_short_inf = (str3ncmp_case_insensitive(src, "inf") | (src[3] ^ '"')) == 0;
+  if(is_short_inf) return true;
+
+  return (str8ncmp_case_insensitive(src, "infinity") | (src[8] ^ '"')) == 0;
+}
+
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src, size_t len) {
+  if (len > 8) { return is_valid_inf_atom(src); }
+  if (len == 8 && str8ncmp_case_insensitive(src, "infinity") == 0) {
+    return true;
+  }
+  if (len > 3) {
+    return (str3ncmp_case_insensitive(src, "inf")
+          | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+  }
+  if (len == 3) { return str3ncmp_case_insensitive(src, "inf") == 0; }
+  return false;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
 } // namespace atomparsing
 } // unnamed namespace
 } // namespace haswell
@@ -19952,7 +22751,7 @@ inline simdjson_warn_unused error_code dom_parser_implementation::set_capacity(s
 }

 inline simdjson_warn_unused error_code dom_parser_implementation::set_max_depth(size_t max_depth) noexcept {
-  if(max_depth == 0 || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
+  if(max_depth == 0 || max_depth > SIMDJSON_MAX_DEPTH || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
   // Stage 2 stacks
   open_containers.reset(new (std::nothrow) open_container[max_depth]);
   is_array.reset(new (std::nothrow) bool[max_depth]);
@@ -20131,6 +22930,7 @@ protected:
 /* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_NUMBERPARSING_H */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/base.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/jsoncharutils.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/atomparsing.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/internal/numberparsing_tables.h" */
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */

@@ -20170,6 +22970,18 @@ simdjson_inline double to_double(uint64_t mantissa, uint64_t real_exponent, bool
     return d;
 }

+// Convert a mantissa, an exponent and a sign bit into an ieee32 float (binary32).
+// The real_exponent needs to be in [0, 254] (technically real_exponent = 255 would be acceptable).
+// The mantissa should be in [0,1<<24). The bit at index (1U << 23) will be zeroed.
+simdjson_inline float to_float(uint32_t mantissa, uint32_t real_exponent, bool negative) {
+    float f;
+    mantissa &= ~(uint32_t(1) << 23);
+    mantissa |= real_exponent << 23;
+    mantissa |= ((static_cast<uint32_t>(negative)) << 31);
+    std::memcpy(&f, &mantissa, sizeof(f));
+    return f;
+}
+
 // Attempts to compute i * 10^(power) exactly; and if "negative" is
 // true, negate the result.
 // This function will only work in some cases, when it does not work, success is
@@ -20426,6 +23238,219 @@ simdjson_inline bool compute_float_64(int64_t power, uint64_t i, bool negative,
   return true;
 }

+// Attempts to compute i * 10^(power) as a binary32 (float) value; and if
+// "negative" is true, negate the result.
+//
+// This is the same Eisel-Lemire algorithm as compute_float_64, retargeted at
+// binary32. It is adapted from fast_float
+// (https://github.com/fastfloat/fast_float), where the routine is written once
+// and instantiated for each binary format. We only need 24 bits of mantissa
+// instead of 53, so we ask the truncated product for 24 + 2 = 26 accurate bits
+// (plus one bit that may be lost to the "upperbit" shift), and we shift the
+// product right by (upperbit + 64 - 23 - 3) instead of (upperbit + 64 - 52 - 3).
+//
+// The power-of-five table is shared with the binary64 code: the useful range
+// for binary32, [smallest_power_binary32, largest_power_binary32], is a strict
+// subset of [smallest_power, largest_power].
+//
+// The function returns false when the result would be infinite: simdjson
+// refuses to parse infinite values, so the caller reports an error. Unlike
+// compute_float_64, a false return never means "try harder": the accuracy of
+// the two-limb product is guaranteed by Noble Mushtak and Daniel Lemire,
+// "Fast Number Parsing Without Fallback" (https://arxiv.org/abs/2212.06644).
+// The caller must still fall back to a slow path when the 64-bit mantissa i
+// was truncated (more than 19 significant digits).
+//
+// We assume that power is in the [smallest_power, largest_power] interval: the
+// caller is responsible for this check.
+simdjson_inline bool compute_float_32(int64_t power, uint64_t i, bool negative, float &d) {
+  // Powers of ten that are exactly representable as binary32 values: 10^k is
+  // exact as long as k <= 10 (5^10 = 9765625 < 2**24 while 5^11 > 2**24).
+  static constexpr float power_of_ten_float[] = {
+      1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+      1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+  // The range of powers of ten that a non-zero, finite binary32 value can be
+  // built from. Anything smaller rounds to zero, anything larger is infinite.
+  // These are the smallest_power_of_ten()/largest_power_of_ten() constants that
+  // fast_float uses for binary32.
+  constexpr int smallest_power_binary32 = -65;
+  constexpr int largest_power_binary32 = 38;
+
+  // We start with the fast path described in
+  // Clinger WD. How to read floating point numbers accurately.
+  // ACM SIGPLAN Notices. 1990
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+  // We cannot be certain that x/y is rounded to nearest.
+  if (0 <= power && power <= 10 && i <= 16777215)
+#else
+  if (-10 <= power && power <= 10 && i <= 16777215)
+#endif
+  {
+    // Convert the integer into a float. This is lossless since
+    // 0 <= i <= 2^24 - 1.
+    d = float(i);
+    // Both d and the power of ten are exactly representable as binary32
+    // values, so the product (or quotient) is correctly rounded.
+    if (power < 0) {
+      d = d / power_of_ten_float[-power];
+    } else {
+      d = d * power_of_ten_float[power];
+    }
+    if (negative) {
+      d = -d;
+    }
+    return true;
+  }
+
+  // The fast path has failed, so we fall back on the Eisel-Lemire algorithm.
+  // It needs i > 0 (so that the leading bit of i can be normalized), so we
+  // handle i == 0 separately. We also handle the powers of ten that are so
+  // small (or so large) that the answer is zero (or infinite) whatever the
+  // mantissa is.
+  if (i == 0 || power < smallest_power_binary32) {
+    d = negative ? -0.0f : 0.0f;
+    return true;
+  }
+  if (power > largest_power_binary32) {
+    // We have, for sure, an infinite value.
+    return false;
+  }
+
+  // The exponent is 128 + 63 + power + floor(log(5**power)/log(2)).
+  // The 128 comes from the ieee32 standard (the minimal exponent is -127).
+  // The 63 comes from the fact that we use a 64-bit word.
+  // See compute_float_64 for a discussion of the magical 152170 + 65536.
+  int64_t exponent = (((152170 + 65536) * power) >> 16) + 128 + 63;
+
+  // We want the most significant bit of i to be 1. Shift if needed.
+  int lz = leading_zeroes(i);
+  i <<= lz;
+
+  // We want the most significant 64 bits of the product i * 5**power. It is
+  // safe to index the table because
+  // smallest_power <= smallest_power_binary32 <= power
+  //               <= largest_power_binary32 <= largest_power.
+  const uint32_t index = 2 * uint32_t(power - simdjson::internal::smallest_power);
+#if SIMDJSON_STATIC_REFLECTION
+  simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index]);
+#else
+  simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index]);
+#endif
+
+  // Unless the least significant 38 bits of the high (64-bit) part of the full
+  // product are all 1s, then we know that the most significant 26 bits are
+  // exact and no further work is needed. Having 26 bits is necessary because
+  // we need 24 bits for the mantissa but we have to have one rounding bit and
+  // we can waste a bit if the most significant bit of the product is zero.
+  // (For binary64, the same reasoning gives 55 bits and a 9-bit mask.)
+  if((firstproduct.high & 0x3FFFFFFFFF) == 0x3FFFFFFFFF) {
+    // The truncated multiplication was not accurate enough; use the next 64
+    // bits of the power of five to refine it. See compute_float_64 for a
+    // detailed discussion.
+#if SIMDJSON_STATIC_REFLECTION
+    simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index + 1]);
+#else
+    simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index + 1]);
+#endif
+    firstproduct.low += secondproduct.high;
+    if(secondproduct.high > firstproduct.low) { firstproduct.high++; }
+  }
+  uint64_t lower = firstproduct.low;
+  uint64_t upper = firstproduct.high;
+  // The final mantissa should be 24 bits with a leading 1.
+  // We shift it so that it occupies 25 bits with a leading 1.
+  ///////
+  uint64_t upperbit = upper >> 63;
+  uint64_t mantissa = upper >> (upperbit + 38); // 38 == 64 - 23 - 3
+  lz += int(1 ^ upperbit);
+
+  // Here we have mantissa < (1<<25).
+  int64_t real_exponent = exponent - lz;
+  if (simdjson_unlikely(real_exponent <= 0)) { // we have a subnormal?
+    // Here we have that real_exponent <= 0 so -real_exponent >= 0
+    if(-real_exponent + 1 >= 64) { // if we have more than 64 bits below the minimum exponent, you have a zero for sure.
+      d = negative ? -0.0f : 0.0f;
+      return true;
+    }
+    // next line is safe because -real_exponent + 1 < 64
+    mantissa >>= -real_exponent + 1;
+    // Thankfully, we can't have both "round-to-even" and subnormals because
+    // "round-to-even" only occurs for powers close to 0.
+    mantissa += (mantissa & 1); // round up
+    mantissa >>= 1;
+    // As in compute_float_64, rounding up may take us out of the subnormal
+    // range, so we can only decide after rounding.
+    real_exponent = (mantissa < (uint64_t(1) << 23)) ? 0 : 1;
+    d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+    return true;
+  }
+  // We have to round to even. The "to even" part is only a problem when we are
+  // right in between two floats, which we guard against. The bounds on the
+  // power of ten are those of fast_float's
+  // min_exponent_round_to_even()/max_exponent_round_to_even() for binary32:
+  // when q >= 0, (2m+1) must be divisible by 5^q with 5^q <= 2^25, so q <= 10;
+  // when q < 0, we need 2^24 x 5^{-q} < 2^{64}, so q >= -17.
+  if (simdjson_unlikely((lower <= 1) && (power >= -17) && (power <= 10) && ((mantissa & 3) == 1))) {
+    if((mantissa << (upperbit + 38)) == upper) {
+      mantissa &= ~uint64_t(1);   // flip it so that we do not round up
+    }
+  }
+
+  mantissa += mantissa & 1;
+  mantissa >>= 1;
+
+  // Here we have mantissa < (1<<24), unless there was an overflow
+  if (mantissa >= (uint64_t(1) << 24)) {
+    mantissa = (uint64_t(1) << 23);
+    real_exponent++;
+  }
+  mantissa &= ~(uint64_t(1) << 23);
+  // we have to check that real_exponent is in range, otherwise we bail out
+  if (simdjson_unlikely(real_exponent > 254)) {
+    // We have an infinite value!!! We could actually throw an error here if we could.
+    return false;
+  }
+  d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+  return true;
+}
+
+#if SIMDJSON_ENABLE_NAN_INF
+// Parses a nan or infinity. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, double& d) noexcept {
+  if (atomparsing::is_valid_inf_atom(src)) {
+    double inf = std::numeric_limits<double>::infinity();
+    d = negative ? -inf : inf;
+    return true;
+  }
+
+  if (atomparsing::is_valid_nan_atom(src)) {
+    d = std::numeric_limits<double>::quiet_NaN();
+    return true;
+  }
+
+  return false;
+}
+
+// Parses a nan or infinity as a binary32 value. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, float& d) noexcept {
+  if (atomparsing::is_valid_inf_atom(src)) {
+    float inf = std::numeric_limits<float>::infinity();
+    d = negative ? -inf : inf;
+    return true;
+  }
+
+  if (atomparsing::is_valid_nan_atom(src)) {
+    d = std::numeric_limits<float>::quiet_NaN();
+    return true;
+  }
+
+  return false;
+}
+#endif
+
 // We call a fallback floating-point parser that might be slow. Note
 // it will accept JSON numbers, but the JSON spec. is more restrictive so
 // before you call parse_float_fallback, you need to have validated the input
@@ -20463,6 +23488,16 @@ static bool parse_float_fallback(const uint8_t *ptr, const uint8_t *end_ptr, dou
   return !(*outDouble > (std::numeric_limits<double>::max)() || *outDouble < std::numeric_limits<double>::lowest());
 }

+// Same as parse_float_fallback, but for binary32 (float) values. Going through
+// the binary64 fallback and then rounding to binary32 would be subject to
+// double rounding, so we run the fallback algorithm directly on binary32.
+static bool parse_float_fallback(const uint8_t *ptr, float *outFloat) {
+  *outFloat = simdjson::internal::from_chars_float(reinterpret_cast<const char *>(ptr));
+  // We do not accept infinite values. See the binary64 version above for why we
+  // do not use std::isfinite.
+  return !(*outFloat > (std::numeric_limits<float>::max)() || *outFloat < std::numeric_limits<float>::lowest());
+}
+
 // check quickly whether the next 8 chars are made of digits
 // at a glance, it looks better than Mula's
 // http://0x80.pl/articles/swar-digits-validate.html
@@ -20481,6 +23516,28 @@ simdjson_inline bool is_made_of_eight_digits_fast(const uint8_t *chars) {
           0x3333333333333333);
 }

+// The same idea for four characters. A block of eight only bites when eight
+// digits are left, so a run of 4 to 7 digits used to fall back to one digit at a
+// time; taking four of them at once is what makes the tail of a long fraction
+// cheap. Adapted from fast_float, credit @aqrit.
+simdjson_inline bool is_made_of_four_digits_fast(const uint8_t *chars) {
+  uint32_t val;
+  // Reads up to 3 bytes beyond the digits, which SIMDJSON_PADDING covers.
+  static_assert(3 <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be bigger than 3");
+  std::memcpy(&val, chars, 4);
+  return !(((val + 0x46464646) | (val - 0x30303030)) & 0x80808080);
+}
+
+// Only call this when is_made_of_four_digits_fast() says the four characters are
+// digits. Little-endian: guarded by SIMDJSON_SWAR_NUMBER_PARSING.
+simdjson_inline uint32_t parse_four_digits_unrolled(const uint8_t *chars) {
+  uint32_t val;
+  std::memcpy(&val, chars, 4);
+  val -= 0x30303030;
+  val = (val * 10) + (val >> 8);
+  return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
 template<typename I>
 SIMDJSON_NO_SANITIZE_UNDEFINED // We deliberately allow overflow here and check later
 simdjson_inline bool parse_digit(const uint8_t c, I &i) {
@@ -20497,26 +23554,73 @@ simdjson_inline bool is_digit(const uint8_t c) {
   return static_cast<uint8_t>(c - '0') <= 9;
 }

-simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
-  // we continue with the fiction that we have an integer. If the
-  // floating point number is representable as x * 10^z for some integer
-  // z that fits in 53 bits, then we will be able to convert back the
-  // the integer into a float in a lossless manner.
-  const uint8_t *const first_after_period = p;
+// Consumes a run of digits into i, eight at a time while eight are available,
+// then four, then one at a time. Overflow is deliberate: the caller counts the
+// digits and falls back when there are too many for a 64-bit mantissa.
+//
+// Long runs of digits are what a float-heavy document is made of: the fraction
+// of a binary64 printed to full precision is 15 to 17 digits, and reading those
+// one at a time is the single largest cost in parse_double(). Integer parts are
+// left alone: they are usually a handful of digits, and there the check for a
+// block of eight is wasted work.
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_fraction_digits(const uint8_t *&p, uint64_t &i) {
+#ifdef SIMDJSON_SWAR_NUMBER_PARSING
+#if SIMDJSON_SWAR_NUMBER_PARSING
+  while (is_made_of_eight_digits_fast(p)) {
+    i = i * 100000000 + parse_eight_digits_unrolled(p);
+    p += 8;
+  }
+  // A 4 to 7 digit remainder is the common case once the blocks of eight are
+  // gone: 15 fraction digits leave 7, and 17 leave 1 after two blocks.
+  if (is_made_of_four_digits_fast(p)) {
+    i = i * 10000 + parse_four_digits_unrolled(p);
+    p += 4;
+  }
+#endif // SIMDJSON_SWAR_NUMBER_PARSING
+#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
+  while (parse_digit(*p, i)) { p++; }
+}

+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_float_integer_digits(const uint8_t *&p, uint64_t &i,
+                                                bool &leading_zero) {
+  if (!parse_digit(*p, i)) { leading_zero = true; return; }
+  p++;
+  leading_zero = (i == 0);
+  while (parse_digit(*p, i)) { p++; }
+}
+
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_integer_digits(const uint8_t *&p, uint64_t &i) {
 #ifdef SIMDJSON_SWAR_NUMBER_PARSING
 #if SIMDJSON_SWAR_NUMBER_PARSING
-  // this helps if we have lots of decimals!
-  // this turns out to be frequent enough.
+  // Identifiers, timestamps and counters often have eight digits or more.
+  // Prior related work: jsonifier parses integers as eight-digit SWAR words
+  // (str_to_i.hpp, https://github.com/nihilai-collective/Jsonifier). This
+  // takes one such word with parse_eight_digits_unrolled, the routine
+  // simdjson uses for long fractions.
   if (is_made_of_eight_digits_fast(p)) {
     i = i * 100000000 + parse_eight_digits_unrolled(p);
     p += 8;
   }
+  const uint8_t *const swar_end = p + 8;
+  while (p < swar_end && is_made_of_four_digits_fast(p)) {
+    i = i * 10000 + parse_four_digits_unrolled(p);
+    p += 4;
+  }
 #endif // SIMDJSON_SWAR_NUMBER_PARSING
 #endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
-  // Unrolling the first digit makes a small difference on some implementations (e.g. westmere)
-  if (parse_digit(*p, i)) { ++p; }
   while (parse_digit(*p, i)) { p++; }
+}
+
+simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
+  // we continue with the fiction that we have an integer. If the
+  // floating point number is representable as x * 10^z for some integer
+  // z that fits in 53 bits, then we will be able to convert back the
+  // the integer into a float in a lossless manner.
+  const uint8_t *const first_after_period = p;
+  parse_fraction_digits(p, i);
   exponent = first_after_period - p;
   // Decimal without digits (123.) is illegal
   if (exponent == 0) {
@@ -20605,7 +23709,7 @@ simdjson_inline size_t significant_digits(const uint8_t * start_digits, size_t d
 } // unnamed namespace

 /** @private */
-static error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
+inline error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
   if (parse_float_fallback(src, answer)) {
     return SUCCESS;
   }
@@ -20688,15 +23792,17 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
   return SUCCESS;              // always succeeds
 }

-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept  { return false; }
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept  { return false; }
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept { return number_type::signed_integer; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline bool is_negative(const uint8_t *) noexcept  { return false; }
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t *) noexcept  { return false; }
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t *) noexcept { return number_type::signed_integer; }
 #else

 // parse the number at src
@@ -20727,7 +23833,21 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
   size_t digit_count = size_t(p - start_digits);
-  if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) { return INVALID_NUMBER(src); }
+  if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // By this point, we know that our input does not begin with a digit. We will attempt
+    // to handle NaN/Infinity.
+
+    double d;
+    if (compute_nan_inf(p, negative, d)) {
+      writer.append_double(d);
+      return SUCCESS;
+    }
+#endif
+
+    return INVALID_NUMBER(src);
+  }

   //
   // Handle floats if there is a . or e (or both)
@@ -20776,7 +23896,7 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
     // - Therefore, if the number is positive and lower than that, it's overflow.
     // - The value we are looking at is less than or equal to INT64_MAX.
     //
-    }  else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return INVALID_NUMBER(src); }
+    }  else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return BIGINT_NUMBER(src); }
   }

   // Write unsigned if it does not fit in a signed integer.
@@ -20875,7 +23995,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const u
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -20973,7 +24093,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_stri
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -21028,7 +24148,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uin
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -21114,7 +24234,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = src;
   uint64_t i = 0;
-  while (parse_digit(*src, i)) { src++; }
+  parse_integer_digits(src, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -21154,11 +24274,20 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   //
   uint64_t i = 0;
   const uint8_t *p = src;
-  p += parse_digit(*p, i);
-  bool leading_zero = (i == 0);
-  while (parse_digit(*p, i)) { p++; }
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
   // no integer digits, or 0123 (zero must be solo)
-  if ( p == src ) { return INCORRECT_TYPE; }
+  if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no loading digits, the number may be nan or infinity.
+    // Attempt to compute those, and return on success.
+    double d;
+    if (compute_nan_inf(p, negative, d)) { return d; }
+#endif
+
+    return INCORRECT_TYPE;
+  }
   if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }

   //
@@ -21169,9 +24298,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   if (simdjson_likely(*p == '.')) {
     p++;
     const uint8_t *start_decimal_digits = p;
-    if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
-    p++;
-    while (parse_digit(*p, i)) { p++; }
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
     exponent = -(p - start_decimal_digits);

     // Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -21220,6 +24348,97 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   return d;
 }

+// Parse a JSON number into a binary32 (float) value.
+//
+// This mirrors parse_double, but it rounds to binary32 directly instead of
+// rounding to binary64 and then to binary32: the latter is subject to double
+// rounding and would not always produce the nearest float.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * src) noexcept {
+  //
+  // Check for minus sign
+  //
+  bool negative = (*src == '-');
+  src += uint8_t(negative);
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, the number may be nan or infinity.
+    // Attempt to compute those, and return on success.
+    float f;
+    if (compute_nan_inf(p, negative, f)) { return f; }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = p-start_digits > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if (*p == 'e' || *p == 'E') {
+    p++;
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while (parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  float d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_32(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
 simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
   return (*src == '-');
 }
@@ -21372,11 +24591,25 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   //
   uint64_t i = 0;
   const uint8_t *p = src;
-  p += parse_digit(*p, i);
-  bool leading_zero = (i == 0);
-  while (parse_digit(*p, i)) { p++; }
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
   // no integer digits, or 0123 (zero must be solo)
-  if ( p == src ) { return INCORRECT_TYPE; }
+  if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, attempt to parse numbers that are either
+    // NaN or Infinity
+    if (atomparsing::is_valid_inf_in_string(src)) {
+      double inf = std::numeric_limits<double>::infinity();
+      return negative ? -inf : inf;
+    }
+
+    if (atomparsing::is_valid_nan_in_string(src)) {
+      return std::numeric_limits<double>::quiet_NaN();
+    }
+#endif
+
+    return INCORRECT_TYPE;
+  }
   if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }

   //
@@ -21387,9 +24620,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   if (simdjson_likely(*p == '.')) {
     p++;
     const uint8_t *start_decimal_digits = p;
-    if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
-    p++;
-    while (parse_digit(*p, i)) { p++; }
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
     exponent = -(p - start_decimal_digits);

     // Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -21438,6 +24670,99 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   return d;
 }

+// Parse a JSON number held inside a JSON string into a binary32 (float) value.
+// See parse_float for why we do not simply round parse_double_in_string.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * src) noexcept {
+  //
+  // Check for minus sign
+  //
+  bool negative = (*(src + 1) == '-');
+  src += uint8_t(negative) + 1;
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, attempt to parse numbers that are either
+    // NaN or Infinity
+    if (atomparsing::is_valid_inf_in_string(src)) {
+      float inf = std::numeric_limits<float>::infinity();
+      return negative ? -inf : inf;
+    }
+
+    if (atomparsing::is_valid_nan_in_string(src)) {
+      return std::numeric_limits<float>::quiet_NaN();
+    }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = p-start_digits > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if (*p == 'e' || *p == 'E') {
+    p++;
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while (parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if (*p != '"') { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  float d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_32(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
 } // unnamed namespace
 #endif // SIMDJSON_SKIPNUMBERPARSING

@@ -21756,16 +25081,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
 }
 #endif

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
-                                uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
-  return _addcarry_u64(0, value1, value2,
-                       reinterpret_cast<unsigned __int64 *>(result));
-#else
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-#endif
-}

 } // unnamed namespace
 } // namespace icelake
@@ -22521,6 +25836,9 @@ namespace atomparsing {
 // to the compile-time constant 1936482662.
 simdjson_inline uint32_t string_to_uint32(const char* str) { uint32_t val; std::memcpy(&val, str, sizeof(uint32_t)); return val; }

+// Acts on the same principle as string_to_uint32, but on an 8-byte block of memory
+simdjson_inline uint64_t string_to_uint64(const char* str) { uint64_t val; std::memcpy(&val, str, sizeof(uint64_t)); return val; }
+

 // Again in str4ncmp we use a memcpy to avoid undefined behavior. The memcpy may appear expensive.
 // Yet all decent optimizing compilers will compile memcpy to a single instruction, just about.
@@ -22532,6 +25850,28 @@ simdjson_inline uint32_t str4ncmp(const uint8_t *src, const char* atom) {
   return srcval ^ string_to_uint32(atom);
 }

+// Checks that the first 8 characters of the input string match the given atom in a case-insensitive manner.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint64_t str8ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+  uint64_t srcval; // we want to avoid unaligned 32-bit loads (undefined in C/C++)
+  static_assert(sizeof(uint64_t) <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be larger than 8 bytes");
+  std::memcpy(&srcval, src, sizeof(uint64_t));
+
+  return (srcval | 0x2020202020202020ull) ^ string_to_uint64(atom);
+}
+
+// Checks that the first 3 characters of 'src' match 'atom' in a case-insensitive way.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint32_t str3ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+  return ((src[0] | 0x20) ^ atom[0]) //
+       | ((src[1] | 0x20) ^ atom[1]) //
+       | ((src[2] | 0x20) ^ atom[2]);
+}
+
 simdjson_warn_unused
 simdjson_inline bool is_valid_true_atom(const uint8_t *src) {
   return (str4ncmp(src, "true") | jsoncharutils::is_not_structural_or_whitespace(src[4])) == 0;
@@ -22568,6 +25908,71 @@ simdjson_inline bool is_valid_null_atom(const uint8_t *src, size_t len) {
   else { return false; }
 }

+#if SIMDJSON_ENABLE_NAN_INF
+// "nan" is 3 bytes; we check characters and then verify the next
+// character is structural or whitespace. We accept both "nan" and "NaN".
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src) {
+  return (str3ncmp_case_insensitive(src, "nan")
+        | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+}
+
+// checks that the next four characters of a string are 'nan"', where the 'nan'
+// is checked in a case-insensitive way.
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_in_string(const uint8_t *src) {
+  return (str3ncmp_case_insensitive(src, "nan") | (src[3] ^ '"')) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src, size_t len) {
+  if (len > 3) { return is_valid_nan_atom(src); }
+  if (len == 3) { return str3ncmp_case_insensitive(src, "nan") == 0; }
+  return false;
+}
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src) {
+  bool is_short_inf = (str3ncmp_case_insensitive(src, "inf")
+                     | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+  if(is_short_inf) return true;
+
+  // Check for 'infinity' (any capitalization)
+  return (str8ncmp_case_insensitive(src, "infinity") | jsoncharutils::is_not_structural_or_whitespace(src[8])) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_in_string(const uint8_t *src) {
+  bool is_short_inf = (str3ncmp_case_insensitive(src, "inf") | (src[3] ^ '"')) == 0;
+  if(is_short_inf) return true;
+
+  return (str8ncmp_case_insensitive(src, "infinity") | (src[8] ^ '"')) == 0;
+}
+
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src, size_t len) {
+  if (len > 8) { return is_valid_inf_atom(src); }
+  if (len == 8 && str8ncmp_case_insensitive(src, "infinity") == 0) {
+    return true;
+  }
+  if (len > 3) {
+    return (str3ncmp_case_insensitive(src, "inf")
+          | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+  }
+  if (len == 3) { return str3ncmp_case_insensitive(src, "inf") == 0; }
+  return false;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
 } // namespace atomparsing
 } // unnamed namespace
 } // namespace icelake
@@ -22658,7 +26063,7 @@ inline simdjson_warn_unused error_code dom_parser_implementation::set_capacity(s
 }

 inline simdjson_warn_unused error_code dom_parser_implementation::set_max_depth(size_t max_depth) noexcept {
-  if(max_depth == 0 || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
+  if(max_depth == 0 || max_depth > SIMDJSON_MAX_DEPTH || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
   // Stage 2 stacks
   open_containers.reset(new (std::nothrow) open_container[max_depth]);
   is_array.reset(new (std::nothrow) bool[max_depth]);
@@ -22837,6 +26242,7 @@ protected:
 /* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_NUMBERPARSING_H */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/base.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/jsoncharutils.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/atomparsing.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/internal/numberparsing_tables.h" */
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */

@@ -22876,6 +26282,18 @@ simdjson_inline double to_double(uint64_t mantissa, uint64_t real_exponent, bool
     return d;
 }

+// Convert a mantissa, an exponent and a sign bit into an ieee32 float (binary32).
+// The real_exponent needs to be in [0, 254] (technically real_exponent = 255 would be acceptable).
+// The mantissa should be in [0,1<<24). The bit at index (1U << 23) will be zeroed.
+simdjson_inline float to_float(uint32_t mantissa, uint32_t real_exponent, bool negative) {
+    float f;
+    mantissa &= ~(uint32_t(1) << 23);
+    mantissa |= real_exponent << 23;
+    mantissa |= ((static_cast<uint32_t>(negative)) << 31);
+    std::memcpy(&f, &mantissa, sizeof(f));
+    return f;
+}
+
 // Attempts to compute i * 10^(power) exactly; and if "negative" is
 // true, negate the result.
 // This function will only work in some cases, when it does not work, success is
@@ -23132,6 +26550,219 @@ simdjson_inline bool compute_float_64(int64_t power, uint64_t i, bool negative,
   return true;
 }

+// Attempts to compute i * 10^(power) as a binary32 (float) value; and if
+// "negative" is true, negate the result.
+//
+// This is the same Eisel-Lemire algorithm as compute_float_64, retargeted at
+// binary32. It is adapted from fast_float
+// (https://github.com/fastfloat/fast_float), where the routine is written once
+// and instantiated for each binary format. We only need 24 bits of mantissa
+// instead of 53, so we ask the truncated product for 24 + 2 = 26 accurate bits
+// (plus one bit that may be lost to the "upperbit" shift), and we shift the
+// product right by (upperbit + 64 - 23 - 3) instead of (upperbit + 64 - 52 - 3).
+//
+// The power-of-five table is shared with the binary64 code: the useful range
+// for binary32, [smallest_power_binary32, largest_power_binary32], is a strict
+// subset of [smallest_power, largest_power].
+//
+// The function returns false when the result would be infinite: simdjson
+// refuses to parse infinite values, so the caller reports an error. Unlike
+// compute_float_64, a false return never means "try harder": the accuracy of
+// the two-limb product is guaranteed by Noble Mushtak and Daniel Lemire,
+// "Fast Number Parsing Without Fallback" (https://arxiv.org/abs/2212.06644).
+// The caller must still fall back to a slow path when the 64-bit mantissa i
+// was truncated (more than 19 significant digits).
+//
+// We assume that power is in the [smallest_power, largest_power] interval: the
+// caller is responsible for this check.
+simdjson_inline bool compute_float_32(int64_t power, uint64_t i, bool negative, float &d) {
+  // Powers of ten that are exactly representable as binary32 values: 10^k is
+  // exact as long as k <= 10 (5^10 = 9765625 < 2**24 while 5^11 > 2**24).
+  static constexpr float power_of_ten_float[] = {
+      1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+      1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+  // The range of powers of ten that a non-zero, finite binary32 value can be
+  // built from. Anything smaller rounds to zero, anything larger is infinite.
+  // These are the smallest_power_of_ten()/largest_power_of_ten() constants that
+  // fast_float uses for binary32.
+  constexpr int smallest_power_binary32 = -65;
+  constexpr int largest_power_binary32 = 38;
+
+  // We start with the fast path described in
+  // Clinger WD. How to read floating point numbers accurately.
+  // ACM SIGPLAN Notices. 1990
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+  // We cannot be certain that x/y is rounded to nearest.
+  if (0 <= power && power <= 10 && i <= 16777215)
+#else
+  if (-10 <= power && power <= 10 && i <= 16777215)
+#endif
+  {
+    // Convert the integer into a float. This is lossless since
+    // 0 <= i <= 2^24 - 1.
+    d = float(i);
+    // Both d and the power of ten are exactly representable as binary32
+    // values, so the product (or quotient) is correctly rounded.
+    if (power < 0) {
+      d = d / power_of_ten_float[-power];
+    } else {
+      d = d * power_of_ten_float[power];
+    }
+    if (negative) {
+      d = -d;
+    }
+    return true;
+  }
+
+  // The fast path has failed, so we fall back on the Eisel-Lemire algorithm.
+  // It needs i > 0 (so that the leading bit of i can be normalized), so we
+  // handle i == 0 separately. We also handle the powers of ten that are so
+  // small (or so large) that the answer is zero (or infinite) whatever the
+  // mantissa is.
+  if (i == 0 || power < smallest_power_binary32) {
+    d = negative ? -0.0f : 0.0f;
+    return true;
+  }
+  if (power > largest_power_binary32) {
+    // We have, for sure, an infinite value.
+    return false;
+  }
+
+  // The exponent is 128 + 63 + power + floor(log(5**power)/log(2)).
+  // The 128 comes from the ieee32 standard (the minimal exponent is -127).
+  // The 63 comes from the fact that we use a 64-bit word.
+  // See compute_float_64 for a discussion of the magical 152170 + 65536.
+  int64_t exponent = (((152170 + 65536) * power) >> 16) + 128 + 63;
+
+  // We want the most significant bit of i to be 1. Shift if needed.
+  int lz = leading_zeroes(i);
+  i <<= lz;
+
+  // We want the most significant 64 bits of the product i * 5**power. It is
+  // safe to index the table because
+  // smallest_power <= smallest_power_binary32 <= power
+  //               <= largest_power_binary32 <= largest_power.
+  const uint32_t index = 2 * uint32_t(power - simdjson::internal::smallest_power);
+#if SIMDJSON_STATIC_REFLECTION
+  simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index]);
+#else
+  simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index]);
+#endif
+
+  // Unless the least significant 38 bits of the high (64-bit) part of the full
+  // product are all 1s, then we know that the most significant 26 bits are
+  // exact and no further work is needed. Having 26 bits is necessary because
+  // we need 24 bits for the mantissa but we have to have one rounding bit and
+  // we can waste a bit if the most significant bit of the product is zero.
+  // (For binary64, the same reasoning gives 55 bits and a 9-bit mask.)
+  if((firstproduct.high & 0x3FFFFFFFFF) == 0x3FFFFFFFFF) {
+    // The truncated multiplication was not accurate enough; use the next 64
+    // bits of the power of five to refine it. See compute_float_64 for a
+    // detailed discussion.
+#if SIMDJSON_STATIC_REFLECTION
+    simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index + 1]);
+#else
+    simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index + 1]);
+#endif
+    firstproduct.low += secondproduct.high;
+    if(secondproduct.high > firstproduct.low) { firstproduct.high++; }
+  }
+  uint64_t lower = firstproduct.low;
+  uint64_t upper = firstproduct.high;
+  // The final mantissa should be 24 bits with a leading 1.
+  // We shift it so that it occupies 25 bits with a leading 1.
+  ///////
+  uint64_t upperbit = upper >> 63;
+  uint64_t mantissa = upper >> (upperbit + 38); // 38 == 64 - 23 - 3
+  lz += int(1 ^ upperbit);
+
+  // Here we have mantissa < (1<<25).
+  int64_t real_exponent = exponent - lz;
+  if (simdjson_unlikely(real_exponent <= 0)) { // we have a subnormal?
+    // Here we have that real_exponent <= 0 so -real_exponent >= 0
+    if(-real_exponent + 1 >= 64) { // if we have more than 64 bits below the minimum exponent, you have a zero for sure.
+      d = negative ? -0.0f : 0.0f;
+      return true;
+    }
+    // next line is safe because -real_exponent + 1 < 64
+    mantissa >>= -real_exponent + 1;
+    // Thankfully, we can't have both "round-to-even" and subnormals because
+    // "round-to-even" only occurs for powers close to 0.
+    mantissa += (mantissa & 1); // round up
+    mantissa >>= 1;
+    // As in compute_float_64, rounding up may take us out of the subnormal
+    // range, so we can only decide after rounding.
+    real_exponent = (mantissa < (uint64_t(1) << 23)) ? 0 : 1;
+    d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+    return true;
+  }
+  // We have to round to even. The "to even" part is only a problem when we are
+  // right in between two floats, which we guard against. The bounds on the
+  // power of ten are those of fast_float's
+  // min_exponent_round_to_even()/max_exponent_round_to_even() for binary32:
+  // when q >= 0, (2m+1) must be divisible by 5^q with 5^q <= 2^25, so q <= 10;
+  // when q < 0, we need 2^24 x 5^{-q} < 2^{64}, so q >= -17.
+  if (simdjson_unlikely((lower <= 1) && (power >= -17) && (power <= 10) && ((mantissa & 3) == 1))) {
+    if((mantissa << (upperbit + 38)) == upper) {
+      mantissa &= ~uint64_t(1);   // flip it so that we do not round up
+    }
+  }
+
+  mantissa += mantissa & 1;
+  mantissa >>= 1;
+
+  // Here we have mantissa < (1<<24), unless there was an overflow
+  if (mantissa >= (uint64_t(1) << 24)) {
+    mantissa = (uint64_t(1) << 23);
+    real_exponent++;
+  }
+  mantissa &= ~(uint64_t(1) << 23);
+  // we have to check that real_exponent is in range, otherwise we bail out
+  if (simdjson_unlikely(real_exponent > 254)) {
+    // We have an infinite value!!! We could actually throw an error here if we could.
+    return false;
+  }
+  d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+  return true;
+}
+
+#if SIMDJSON_ENABLE_NAN_INF
+// Parses a nan or infinity. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, double& d) noexcept {
+  if (atomparsing::is_valid_inf_atom(src)) {
+    double inf = std::numeric_limits<double>::infinity();
+    d = negative ? -inf : inf;
+    return true;
+  }
+
+  if (atomparsing::is_valid_nan_atom(src)) {
+    d = std::numeric_limits<double>::quiet_NaN();
+    return true;
+  }
+
+  return false;
+}
+
+// Parses a nan or infinity as a binary32 value. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, float& d) noexcept {
+  if (atomparsing::is_valid_inf_atom(src)) {
+    float inf = std::numeric_limits<float>::infinity();
+    d = negative ? -inf : inf;
+    return true;
+  }
+
+  if (atomparsing::is_valid_nan_atom(src)) {
+    d = std::numeric_limits<float>::quiet_NaN();
+    return true;
+  }
+
+  return false;
+}
+#endif
+
 // We call a fallback floating-point parser that might be slow. Note
 // it will accept JSON numbers, but the JSON spec. is more restrictive so
 // before you call parse_float_fallback, you need to have validated the input
@@ -23169,6 +26800,16 @@ static bool parse_float_fallback(const uint8_t *ptr, const uint8_t *end_ptr, dou
   return !(*outDouble > (std::numeric_limits<double>::max)() || *outDouble < std::numeric_limits<double>::lowest());
 }

+// Same as parse_float_fallback, but for binary32 (float) values. Going through
+// the binary64 fallback and then rounding to binary32 would be subject to
+// double rounding, so we run the fallback algorithm directly on binary32.
+static bool parse_float_fallback(const uint8_t *ptr, float *outFloat) {
+  *outFloat = simdjson::internal::from_chars_float(reinterpret_cast<const char *>(ptr));
+  // We do not accept infinite values. See the binary64 version above for why we
+  // do not use std::isfinite.
+  return !(*outFloat > (std::numeric_limits<float>::max)() || *outFloat < std::numeric_limits<float>::lowest());
+}
+
 // check quickly whether the next 8 chars are made of digits
 // at a glance, it looks better than Mula's
 // http://0x80.pl/articles/swar-digits-validate.html
@@ -23187,6 +26828,28 @@ simdjson_inline bool is_made_of_eight_digits_fast(const uint8_t *chars) {
           0x3333333333333333);
 }

+// The same idea for four characters. A block of eight only bites when eight
+// digits are left, so a run of 4 to 7 digits used to fall back to one digit at a
+// time; taking four of them at once is what makes the tail of a long fraction
+// cheap. Adapted from fast_float, credit @aqrit.
+simdjson_inline bool is_made_of_four_digits_fast(const uint8_t *chars) {
+  uint32_t val;
+  // Reads up to 3 bytes beyond the digits, which SIMDJSON_PADDING covers.
+  static_assert(3 <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be bigger than 3");
+  std::memcpy(&val, chars, 4);
+  return !(((val + 0x46464646) | (val - 0x30303030)) & 0x80808080);
+}
+
+// Only call this when is_made_of_four_digits_fast() says the four characters are
+// digits. Little-endian: guarded by SIMDJSON_SWAR_NUMBER_PARSING.
+simdjson_inline uint32_t parse_four_digits_unrolled(const uint8_t *chars) {
+  uint32_t val;
+  std::memcpy(&val, chars, 4);
+  val -= 0x30303030;
+  val = (val * 10) + (val >> 8);
+  return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
 template<typename I>
 SIMDJSON_NO_SANITIZE_UNDEFINED // We deliberately allow overflow here and check later
 simdjson_inline bool parse_digit(const uint8_t c, I &i) {
@@ -23203,26 +26866,73 @@ simdjson_inline bool is_digit(const uint8_t c) {
   return static_cast<uint8_t>(c - '0') <= 9;
 }

-simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
-  // we continue with the fiction that we have an integer. If the
-  // floating point number is representable as x * 10^z for some integer
-  // z that fits in 53 bits, then we will be able to convert back the
-  // the integer into a float in a lossless manner.
-  const uint8_t *const first_after_period = p;
+// Consumes a run of digits into i, eight at a time while eight are available,
+// then four, then one at a time. Overflow is deliberate: the caller counts the
+// digits and falls back when there are too many for a 64-bit mantissa.
+//
+// Long runs of digits are what a float-heavy document is made of: the fraction
+// of a binary64 printed to full precision is 15 to 17 digits, and reading those
+// one at a time is the single largest cost in parse_double(). Integer parts are
+// left alone: they are usually a handful of digits, and there the check for a
+// block of eight is wasted work.
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_fraction_digits(const uint8_t *&p, uint64_t &i) {
+#ifdef SIMDJSON_SWAR_NUMBER_PARSING
+#if SIMDJSON_SWAR_NUMBER_PARSING
+  while (is_made_of_eight_digits_fast(p)) {
+    i = i * 100000000 + parse_eight_digits_unrolled(p);
+    p += 8;
+  }
+  // A 4 to 7 digit remainder is the common case once the blocks of eight are
+  // gone: 15 fraction digits leave 7, and 17 leave 1 after two blocks.
+  if (is_made_of_four_digits_fast(p)) {
+    i = i * 10000 + parse_four_digits_unrolled(p);
+    p += 4;
+  }
+#endif // SIMDJSON_SWAR_NUMBER_PARSING
+#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
+  while (parse_digit(*p, i)) { p++; }
+}
+
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_float_integer_digits(const uint8_t *&p, uint64_t &i,
+                                                bool &leading_zero) {
+  if (!parse_digit(*p, i)) { leading_zero = true; return; }
+  p++;
+  leading_zero = (i == 0);
+  while (parse_digit(*p, i)) { p++; }
+}

+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_integer_digits(const uint8_t *&p, uint64_t &i) {
 #ifdef SIMDJSON_SWAR_NUMBER_PARSING
 #if SIMDJSON_SWAR_NUMBER_PARSING
-  // this helps if we have lots of decimals!
-  // this turns out to be frequent enough.
+  // Identifiers, timestamps and counters often have eight digits or more.
+  // Prior related work: jsonifier parses integers as eight-digit SWAR words
+  // (str_to_i.hpp, https://github.com/nihilai-collective/Jsonifier). This
+  // takes one such word with parse_eight_digits_unrolled, the routine
+  // simdjson uses for long fractions.
   if (is_made_of_eight_digits_fast(p)) {
     i = i * 100000000 + parse_eight_digits_unrolled(p);
     p += 8;
   }
+  const uint8_t *const swar_end = p + 8;
+  while (p < swar_end && is_made_of_four_digits_fast(p)) {
+    i = i * 10000 + parse_four_digits_unrolled(p);
+    p += 4;
+  }
 #endif // SIMDJSON_SWAR_NUMBER_PARSING
 #endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
-  // Unrolling the first digit makes a small difference on some implementations (e.g. westmere)
-  if (parse_digit(*p, i)) { ++p; }
   while (parse_digit(*p, i)) { p++; }
+}
+
+simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
+  // we continue with the fiction that we have an integer. If the
+  // floating point number is representable as x * 10^z for some integer
+  // z that fits in 53 bits, then we will be able to convert back the
+  // the integer into a float in a lossless manner.
+  const uint8_t *const first_after_period = p;
+  parse_fraction_digits(p, i);
   exponent = first_after_period - p;
   // Decimal without digits (123.) is illegal
   if (exponent == 0) {
@@ -23311,7 +27021,7 @@ simdjson_inline size_t significant_digits(const uint8_t * start_digits, size_t d
 } // unnamed namespace

 /** @private */
-static error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
+inline error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
   if (parse_float_fallback(src, answer)) {
     return SUCCESS;
   }
@@ -23394,15 +27104,17 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
   return SUCCESS;              // always succeeds
 }

-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept  { return false; }
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept  { return false; }
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept { return number_type::signed_integer; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline bool is_negative(const uint8_t *) noexcept  { return false; }
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t *) noexcept  { return false; }
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t *) noexcept { return number_type::signed_integer; }
 #else

 // parse the number at src
@@ -23433,7 +27145,21 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
   size_t digit_count = size_t(p - start_digits);
-  if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) { return INVALID_NUMBER(src); }
+  if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // By this point, we know that our input does not begin with a digit. We will attempt
+    // to handle NaN/Infinity.
+
+    double d;
+    if (compute_nan_inf(p, negative, d)) {
+      writer.append_double(d);
+      return SUCCESS;
+    }
+#endif
+
+    return INVALID_NUMBER(src);
+  }

   //
   // Handle floats if there is a . or e (or both)
@@ -23482,7 +27208,7 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
     // - Therefore, if the number is positive and lower than that, it's overflow.
     // - The value we are looking at is less than or equal to INT64_MAX.
     //
-    }  else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return INVALID_NUMBER(src); }
+    }  else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return BIGINT_NUMBER(src); }
   }

   // Write unsigned if it does not fit in a signed integer.
@@ -23581,7 +27307,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const u
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -23679,7 +27405,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_stri
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -23734,7 +27460,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uin
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -23820,7 +27546,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = src;
   uint64_t i = 0;
-  while (parse_digit(*src, i)) { src++; }
+  parse_integer_digits(src, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -23860,11 +27586,20 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   //
   uint64_t i = 0;
   const uint8_t *p = src;
-  p += parse_digit(*p, i);
-  bool leading_zero = (i == 0);
-  while (parse_digit(*p, i)) { p++; }
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
   // no integer digits, or 0123 (zero must be solo)
-  if ( p == src ) { return INCORRECT_TYPE; }
+  if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no loading digits, the number may be nan or infinity.
+    // Attempt to compute those, and return on success.
+    double d;
+    if (compute_nan_inf(p, negative, d)) { return d; }
+#endif
+
+    return INCORRECT_TYPE;
+  }
   if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }

   //
@@ -23875,9 +27610,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   if (simdjson_likely(*p == '.')) {
     p++;
     const uint8_t *start_decimal_digits = p;
-    if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
-    p++;
-    while (parse_digit(*p, i)) { p++; }
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
     exponent = -(p - start_decimal_digits);

     // Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -23926,6 +27660,97 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   return d;
 }

+// Parse a JSON number into a binary32 (float) value.
+//
+// This mirrors parse_double, but it rounds to binary32 directly instead of
+// rounding to binary64 and then to binary32: the latter is subject to double
+// rounding and would not always produce the nearest float.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * src) noexcept {
+  //
+  // Check for minus sign
+  //
+  bool negative = (*src == '-');
+  src += uint8_t(negative);
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, the number may be nan or infinity.
+    // Attempt to compute those, and return on success.
+    float f;
+    if (compute_nan_inf(p, negative, f)) { return f; }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = p-start_digits > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if (*p == 'e' || *p == 'E') {
+    p++;
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while (parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  float d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_32(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
 simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
   return (*src == '-');
 }
@@ -24078,11 +27903,25 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   //
   uint64_t i = 0;
   const uint8_t *p = src;
-  p += parse_digit(*p, i);
-  bool leading_zero = (i == 0);
-  while (parse_digit(*p, i)) { p++; }
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
   // no integer digits, or 0123 (zero must be solo)
-  if ( p == src ) { return INCORRECT_TYPE; }
+  if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, attempt to parse numbers that are either
+    // NaN or Infinity
+    if (atomparsing::is_valid_inf_in_string(src)) {
+      double inf = std::numeric_limits<double>::infinity();
+      return negative ? -inf : inf;
+    }
+
+    if (atomparsing::is_valid_nan_in_string(src)) {
+      return std::numeric_limits<double>::quiet_NaN();
+    }
+#endif
+
+    return INCORRECT_TYPE;
+  }
   if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }

   //
@@ -24093,9 +27932,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   if (simdjson_likely(*p == '.')) {
     p++;
     const uint8_t *start_decimal_digits = p;
-    if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
-    p++;
-    while (parse_digit(*p, i)) { p++; }
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
     exponent = -(p - start_decimal_digits);

     // Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -24144,6 +27982,99 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   return d;
 }

+// Parse a JSON number held inside a JSON string into a binary32 (float) value.
+// See parse_float for why we do not simply round parse_double_in_string.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * src) noexcept {
+  //
+  // Check for minus sign
+  //
+  bool negative = (*(src + 1) == '-');
+  src += uint8_t(negative) + 1;
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, attempt to parse numbers that are either
+    // NaN or Infinity
+    if (atomparsing::is_valid_inf_in_string(src)) {
+      float inf = std::numeric_limits<float>::infinity();
+      return negative ? -inf : inf;
+    }
+
+    if (atomparsing::is_valid_nan_in_string(src)) {
+      return std::numeric_limits<float>::quiet_NaN();
+    }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = p-start_digits > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if (*p == 'e' || *p == 'E') {
+    p++;
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while (parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if (*p != '"') { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  float d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_32(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
 } // unnamed namespace
 #endif // SIMDJSON_SKIPNUMBERPARSING

@@ -24434,16 +28365,6 @@ simdjson_inline int count_ones(uint64_t input_num) {
 }
 #endif

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
-                                         uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
-  *result = value1 + value2;
-  return *result < value1;
-#else
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-#endif
-}

 } // unnamed namespace
 } // namespace ppc64
@@ -25342,6 +29263,9 @@ namespace atomparsing {
 // to the compile-time constant 1936482662.
 simdjson_inline uint32_t string_to_uint32(const char* str) { uint32_t val; std::memcpy(&val, str, sizeof(uint32_t)); return val; }

+// Acts on the same principle as string_to_uint32, but on an 8-byte block of memory
+simdjson_inline uint64_t string_to_uint64(const char* str) { uint64_t val; std::memcpy(&val, str, sizeof(uint64_t)); return val; }
+

 // Again in str4ncmp we use a memcpy to avoid undefined behavior. The memcpy may appear expensive.
 // Yet all decent optimizing compilers will compile memcpy to a single instruction, just about.
@@ -25353,6 +29277,28 @@ simdjson_inline uint32_t str4ncmp(const uint8_t *src, const char* atom) {
   return srcval ^ string_to_uint32(atom);
 }

+// Checks that the first 8 characters of the input string match the given atom in a case-insensitive manner.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint64_t str8ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+  uint64_t srcval; // we want to avoid unaligned 32-bit loads (undefined in C/C++)
+  static_assert(sizeof(uint64_t) <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be larger than 8 bytes");
+  std::memcpy(&srcval, src, sizeof(uint64_t));
+
+  return (srcval | 0x2020202020202020ull) ^ string_to_uint64(atom);
+}
+
+// Checks that the first 3 characters of 'src' match 'atom' in a case-insensitive way.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint32_t str3ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+  return ((src[0] | 0x20) ^ atom[0]) //
+       | ((src[1] | 0x20) ^ atom[1]) //
+       | ((src[2] | 0x20) ^ atom[2]);
+}
+
 simdjson_warn_unused
 simdjson_inline bool is_valid_true_atom(const uint8_t *src) {
   return (str4ncmp(src, "true") | jsoncharutils::is_not_structural_or_whitespace(src[4])) == 0;
@@ -25389,6 +29335,71 @@ simdjson_inline bool is_valid_null_atom(const uint8_t *src, size_t len) {
   else { return false; }
 }

+#if SIMDJSON_ENABLE_NAN_INF
+// "nan" is 3 bytes; we check characters and then verify the next
+// character is structural or whitespace. We accept both "nan" and "NaN".
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src) {
+  return (str3ncmp_case_insensitive(src, "nan")
+        | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+}
+
+// checks that the next four characters of a string are 'nan"', where the 'nan'
+// is checked in a case-insensitive way.
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_in_string(const uint8_t *src) {
+  return (str3ncmp_case_insensitive(src, "nan") | (src[3] ^ '"')) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src, size_t len) {
+  if (len > 3) { return is_valid_nan_atom(src); }
+  if (len == 3) { return str3ncmp_case_insensitive(src, "nan") == 0; }
+  return false;
+}
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src) {
+  bool is_short_inf = (str3ncmp_case_insensitive(src, "inf")
+                     | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+  if(is_short_inf) return true;
+
+  // Check for 'infinity' (any capitalization)
+  return (str8ncmp_case_insensitive(src, "infinity") | jsoncharutils::is_not_structural_or_whitespace(src[8])) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_in_string(const uint8_t *src) {
+  bool is_short_inf = (str3ncmp_case_insensitive(src, "inf") | (src[3] ^ '"')) == 0;
+  if(is_short_inf) return true;
+
+  return (str8ncmp_case_insensitive(src, "infinity") | (src[8] ^ '"')) == 0;
+}
+
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src, size_t len) {
+  if (len > 8) { return is_valid_inf_atom(src); }
+  if (len == 8 && str8ncmp_case_insensitive(src, "infinity") == 0) {
+    return true;
+  }
+  if (len > 3) {
+    return (str3ncmp_case_insensitive(src, "inf")
+          | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+  }
+  if (len == 3) { return str3ncmp_case_insensitive(src, "inf") == 0; }
+  return false;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
 } // namespace atomparsing
 } // unnamed namespace
 } // namespace ppc64
@@ -25479,7 +29490,7 @@ inline simdjson_warn_unused error_code dom_parser_implementation::set_capacity(s
 }

 inline simdjson_warn_unused error_code dom_parser_implementation::set_max_depth(size_t max_depth) noexcept {
-  if(max_depth == 0 || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
+  if(max_depth == 0 || max_depth > SIMDJSON_MAX_DEPTH || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
   // Stage 2 stacks
   open_containers.reset(new (std::nothrow) open_container[max_depth]);
   is_array.reset(new (std::nothrow) bool[max_depth]);
@@ -25658,6 +29669,7 @@ protected:
 /* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_NUMBERPARSING_H */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/base.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/jsoncharutils.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/atomparsing.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/internal/numberparsing_tables.h" */
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */

@@ -25697,6 +29709,18 @@ simdjson_inline double to_double(uint64_t mantissa, uint64_t real_exponent, bool
     return d;
 }

+// Convert a mantissa, an exponent and a sign bit into an ieee32 float (binary32).
+// The real_exponent needs to be in [0, 254] (technically real_exponent = 255 would be acceptable).
+// The mantissa should be in [0,1<<24). The bit at index (1U << 23) will be zeroed.
+simdjson_inline float to_float(uint32_t mantissa, uint32_t real_exponent, bool negative) {
+    float f;
+    mantissa &= ~(uint32_t(1) << 23);
+    mantissa |= real_exponent << 23;
+    mantissa |= ((static_cast<uint32_t>(negative)) << 31);
+    std::memcpy(&f, &mantissa, sizeof(f));
+    return f;
+}
+
 // Attempts to compute i * 10^(power) exactly; and if "negative" is
 // true, negate the result.
 // This function will only work in some cases, when it does not work, success is
@@ -25953,6 +29977,219 @@ simdjson_inline bool compute_float_64(int64_t power, uint64_t i, bool negative,
   return true;
 }

+// Attempts to compute i * 10^(power) as a binary32 (float) value; and if
+// "negative" is true, negate the result.
+//
+// This is the same Eisel-Lemire algorithm as compute_float_64, retargeted at
+// binary32. It is adapted from fast_float
+// (https://github.com/fastfloat/fast_float), where the routine is written once
+// and instantiated for each binary format. We only need 24 bits of mantissa
+// instead of 53, so we ask the truncated product for 24 + 2 = 26 accurate bits
+// (plus one bit that may be lost to the "upperbit" shift), and we shift the
+// product right by (upperbit + 64 - 23 - 3) instead of (upperbit + 64 - 52 - 3).
+//
+// The power-of-five table is shared with the binary64 code: the useful range
+// for binary32, [smallest_power_binary32, largest_power_binary32], is a strict
+// subset of [smallest_power, largest_power].
+//
+// The function returns false when the result would be infinite: simdjson
+// refuses to parse infinite values, so the caller reports an error. Unlike
+// compute_float_64, a false return never means "try harder": the accuracy of
+// the two-limb product is guaranteed by Noble Mushtak and Daniel Lemire,
+// "Fast Number Parsing Without Fallback" (https://arxiv.org/abs/2212.06644).
+// The caller must still fall back to a slow path when the 64-bit mantissa i
+// was truncated (more than 19 significant digits).
+//
+// We assume that power is in the [smallest_power, largest_power] interval: the
+// caller is responsible for this check.
+simdjson_inline bool compute_float_32(int64_t power, uint64_t i, bool negative, float &d) {
+  // Powers of ten that are exactly representable as binary32 values: 10^k is
+  // exact as long as k <= 10 (5^10 = 9765625 < 2**24 while 5^11 > 2**24).
+  static constexpr float power_of_ten_float[] = {
+      1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+      1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+  // The range of powers of ten that a non-zero, finite binary32 value can be
+  // built from. Anything smaller rounds to zero, anything larger is infinite.
+  // These are the smallest_power_of_ten()/largest_power_of_ten() constants that
+  // fast_float uses for binary32.
+  constexpr int smallest_power_binary32 = -65;
+  constexpr int largest_power_binary32 = 38;
+
+  // We start with the fast path described in
+  // Clinger WD. How to read floating point numbers accurately.
+  // ACM SIGPLAN Notices. 1990
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+  // We cannot be certain that x/y is rounded to nearest.
+  if (0 <= power && power <= 10 && i <= 16777215)
+#else
+  if (-10 <= power && power <= 10 && i <= 16777215)
+#endif
+  {
+    // Convert the integer into a float. This is lossless since
+    // 0 <= i <= 2^24 - 1.
+    d = float(i);
+    // Both d and the power of ten are exactly representable as binary32
+    // values, so the product (or quotient) is correctly rounded.
+    if (power < 0) {
+      d = d / power_of_ten_float[-power];
+    } else {
+      d = d * power_of_ten_float[power];
+    }
+    if (negative) {
+      d = -d;
+    }
+    return true;
+  }
+
+  // The fast path has failed, so we fall back on the Eisel-Lemire algorithm.
+  // It needs i > 0 (so that the leading bit of i can be normalized), so we
+  // handle i == 0 separately. We also handle the powers of ten that are so
+  // small (or so large) that the answer is zero (or infinite) whatever the
+  // mantissa is.
+  if (i == 0 || power < smallest_power_binary32) {
+    d = negative ? -0.0f : 0.0f;
+    return true;
+  }
+  if (power > largest_power_binary32) {
+    // We have, for sure, an infinite value.
+    return false;
+  }
+
+  // The exponent is 128 + 63 + power + floor(log(5**power)/log(2)).
+  // The 128 comes from the ieee32 standard (the minimal exponent is -127).
+  // The 63 comes from the fact that we use a 64-bit word.
+  // See compute_float_64 for a discussion of the magical 152170 + 65536.
+  int64_t exponent = (((152170 + 65536) * power) >> 16) + 128 + 63;
+
+  // We want the most significant bit of i to be 1. Shift if needed.
+  int lz = leading_zeroes(i);
+  i <<= lz;
+
+  // We want the most significant 64 bits of the product i * 5**power. It is
+  // safe to index the table because
+  // smallest_power <= smallest_power_binary32 <= power
+  //               <= largest_power_binary32 <= largest_power.
+  const uint32_t index = 2 * uint32_t(power - simdjson::internal::smallest_power);
+#if SIMDJSON_STATIC_REFLECTION
+  simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index]);
+#else
+  simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index]);
+#endif
+
+  // Unless the least significant 38 bits of the high (64-bit) part of the full
+  // product are all 1s, then we know that the most significant 26 bits are
+  // exact and no further work is needed. Having 26 bits is necessary because
+  // we need 24 bits for the mantissa but we have to have one rounding bit and
+  // we can waste a bit if the most significant bit of the product is zero.
+  // (For binary64, the same reasoning gives 55 bits and a 9-bit mask.)
+  if((firstproduct.high & 0x3FFFFFFFFF) == 0x3FFFFFFFFF) {
+    // The truncated multiplication was not accurate enough; use the next 64
+    // bits of the power of five to refine it. See compute_float_64 for a
+    // detailed discussion.
+#if SIMDJSON_STATIC_REFLECTION
+    simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index + 1]);
+#else
+    simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index + 1]);
+#endif
+    firstproduct.low += secondproduct.high;
+    if(secondproduct.high > firstproduct.low) { firstproduct.high++; }
+  }
+  uint64_t lower = firstproduct.low;
+  uint64_t upper = firstproduct.high;
+  // The final mantissa should be 24 bits with a leading 1.
+  // We shift it so that it occupies 25 bits with a leading 1.
+  ///////
+  uint64_t upperbit = upper >> 63;
+  uint64_t mantissa = upper >> (upperbit + 38); // 38 == 64 - 23 - 3
+  lz += int(1 ^ upperbit);
+
+  // Here we have mantissa < (1<<25).
+  int64_t real_exponent = exponent - lz;
+  if (simdjson_unlikely(real_exponent <= 0)) { // we have a subnormal?
+    // Here we have that real_exponent <= 0 so -real_exponent >= 0
+    if(-real_exponent + 1 >= 64) { // if we have more than 64 bits below the minimum exponent, you have a zero for sure.
+      d = negative ? -0.0f : 0.0f;
+      return true;
+    }
+    // next line is safe because -real_exponent + 1 < 64
+    mantissa >>= -real_exponent + 1;
+    // Thankfully, we can't have both "round-to-even" and subnormals because
+    // "round-to-even" only occurs for powers close to 0.
+    mantissa += (mantissa & 1); // round up
+    mantissa >>= 1;
+    // As in compute_float_64, rounding up may take us out of the subnormal
+    // range, so we can only decide after rounding.
+    real_exponent = (mantissa < (uint64_t(1) << 23)) ? 0 : 1;
+    d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+    return true;
+  }
+  // We have to round to even. The "to even" part is only a problem when we are
+  // right in between two floats, which we guard against. The bounds on the
+  // power of ten are those of fast_float's
+  // min_exponent_round_to_even()/max_exponent_round_to_even() for binary32:
+  // when q >= 0, (2m+1) must be divisible by 5^q with 5^q <= 2^25, so q <= 10;
+  // when q < 0, we need 2^24 x 5^{-q} < 2^{64}, so q >= -17.
+  if (simdjson_unlikely((lower <= 1) && (power >= -17) && (power <= 10) && ((mantissa & 3) == 1))) {
+    if((mantissa << (upperbit + 38)) == upper) {
+      mantissa &= ~uint64_t(1);   // flip it so that we do not round up
+    }
+  }
+
+  mantissa += mantissa & 1;
+  mantissa >>= 1;
+
+  // Here we have mantissa < (1<<24), unless there was an overflow
+  if (mantissa >= (uint64_t(1) << 24)) {
+    mantissa = (uint64_t(1) << 23);
+    real_exponent++;
+  }
+  mantissa &= ~(uint64_t(1) << 23);
+  // we have to check that real_exponent is in range, otherwise we bail out
+  if (simdjson_unlikely(real_exponent > 254)) {
+    // We have an infinite value!!! We could actually throw an error here if we could.
+    return false;
+  }
+  d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+  return true;
+}
+
+#if SIMDJSON_ENABLE_NAN_INF
+// Parses a nan or infinity. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, double& d) noexcept {
+  if (atomparsing::is_valid_inf_atom(src)) {
+    double inf = std::numeric_limits<double>::infinity();
+    d = negative ? -inf : inf;
+    return true;
+  }
+
+  if (atomparsing::is_valid_nan_atom(src)) {
+    d = std::numeric_limits<double>::quiet_NaN();
+    return true;
+  }
+
+  return false;
+}
+
+// Parses a nan or infinity as a binary32 value. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, float& d) noexcept {
+  if (atomparsing::is_valid_inf_atom(src)) {
+    float inf = std::numeric_limits<float>::infinity();
+    d = negative ? -inf : inf;
+    return true;
+  }
+
+  if (atomparsing::is_valid_nan_atom(src)) {
+    d = std::numeric_limits<float>::quiet_NaN();
+    return true;
+  }
+
+  return false;
+}
+#endif
+
 // We call a fallback floating-point parser that might be slow. Note
 // it will accept JSON numbers, but the JSON spec. is more restrictive so
 // before you call parse_float_fallback, you need to have validated the input
@@ -25990,6 +30227,16 @@ static bool parse_float_fallback(const uint8_t *ptr, const uint8_t *end_ptr, dou
   return !(*outDouble > (std::numeric_limits<double>::max)() || *outDouble < std::numeric_limits<double>::lowest());
 }

+// Same as parse_float_fallback, but for binary32 (float) values. Going through
+// the binary64 fallback and then rounding to binary32 would be subject to
+// double rounding, so we run the fallback algorithm directly on binary32.
+static bool parse_float_fallback(const uint8_t *ptr, float *outFloat) {
+  *outFloat = simdjson::internal::from_chars_float(reinterpret_cast<const char *>(ptr));
+  // We do not accept infinite values. See the binary64 version above for why we
+  // do not use std::isfinite.
+  return !(*outFloat > (std::numeric_limits<float>::max)() || *outFloat < std::numeric_limits<float>::lowest());
+}
+
 // check quickly whether the next 8 chars are made of digits
 // at a glance, it looks better than Mula's
 // http://0x80.pl/articles/swar-digits-validate.html
@@ -26008,6 +30255,28 @@ simdjson_inline bool is_made_of_eight_digits_fast(const uint8_t *chars) {
           0x3333333333333333);
 }

+// The same idea for four characters. A block of eight only bites when eight
+// digits are left, so a run of 4 to 7 digits used to fall back to one digit at a
+// time; taking four of them at once is what makes the tail of a long fraction
+// cheap. Adapted from fast_float, credit @aqrit.
+simdjson_inline bool is_made_of_four_digits_fast(const uint8_t *chars) {
+  uint32_t val;
+  // Reads up to 3 bytes beyond the digits, which SIMDJSON_PADDING covers.
+  static_assert(3 <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be bigger than 3");
+  std::memcpy(&val, chars, 4);
+  return !(((val + 0x46464646) | (val - 0x30303030)) & 0x80808080);
+}
+
+// Only call this when is_made_of_four_digits_fast() says the four characters are
+// digits. Little-endian: guarded by SIMDJSON_SWAR_NUMBER_PARSING.
+simdjson_inline uint32_t parse_four_digits_unrolled(const uint8_t *chars) {
+  uint32_t val;
+  std::memcpy(&val, chars, 4);
+  val -= 0x30303030;
+  val = (val * 10) + (val >> 8);
+  return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
 template<typename I>
 SIMDJSON_NO_SANITIZE_UNDEFINED // We deliberately allow overflow here and check later
 simdjson_inline bool parse_digit(const uint8_t c, I &i) {
@@ -26024,26 +30293,73 @@ simdjson_inline bool is_digit(const uint8_t c) {
   return static_cast<uint8_t>(c - '0') <= 9;
 }

-simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
-  // we continue with the fiction that we have an integer. If the
-  // floating point number is representable as x * 10^z for some integer
-  // z that fits in 53 bits, then we will be able to convert back the
-  // the integer into a float in a lossless manner.
-  const uint8_t *const first_after_period = p;
+// Consumes a run of digits into i, eight at a time while eight are available,
+// then four, then one at a time. Overflow is deliberate: the caller counts the
+// digits and falls back when there are too many for a 64-bit mantissa.
+//
+// Long runs of digits are what a float-heavy document is made of: the fraction
+// of a binary64 printed to full precision is 15 to 17 digits, and reading those
+// one at a time is the single largest cost in parse_double(). Integer parts are
+// left alone: they are usually a handful of digits, and there the check for a
+// block of eight is wasted work.
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_fraction_digits(const uint8_t *&p, uint64_t &i) {
+#ifdef SIMDJSON_SWAR_NUMBER_PARSING
+#if SIMDJSON_SWAR_NUMBER_PARSING
+  while (is_made_of_eight_digits_fast(p)) {
+    i = i * 100000000 + parse_eight_digits_unrolled(p);
+    p += 8;
+  }
+  // A 4 to 7 digit remainder is the common case once the blocks of eight are
+  // gone: 15 fraction digits leave 7, and 17 leave 1 after two blocks.
+  if (is_made_of_four_digits_fast(p)) {
+    i = i * 10000 + parse_four_digits_unrolled(p);
+    p += 4;
+  }
+#endif // SIMDJSON_SWAR_NUMBER_PARSING
+#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
+  while (parse_digit(*p, i)) { p++; }
+}
+
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_float_integer_digits(const uint8_t *&p, uint64_t &i,
+                                                bool &leading_zero) {
+  if (!parse_digit(*p, i)) { leading_zero = true; return; }
+  p++;
+  leading_zero = (i == 0);
+  while (parse_digit(*p, i)) { p++; }
+}

+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_integer_digits(const uint8_t *&p, uint64_t &i) {
 #ifdef SIMDJSON_SWAR_NUMBER_PARSING
 #if SIMDJSON_SWAR_NUMBER_PARSING
-  // this helps if we have lots of decimals!
-  // this turns out to be frequent enough.
+  // Identifiers, timestamps and counters often have eight digits or more.
+  // Prior related work: jsonifier parses integers as eight-digit SWAR words
+  // (str_to_i.hpp, https://github.com/nihilai-collective/Jsonifier). This
+  // takes one such word with parse_eight_digits_unrolled, the routine
+  // simdjson uses for long fractions.
   if (is_made_of_eight_digits_fast(p)) {
     i = i * 100000000 + parse_eight_digits_unrolled(p);
     p += 8;
   }
+  const uint8_t *const swar_end = p + 8;
+  while (p < swar_end && is_made_of_four_digits_fast(p)) {
+    i = i * 10000 + parse_four_digits_unrolled(p);
+    p += 4;
+  }
 #endif // SIMDJSON_SWAR_NUMBER_PARSING
 #endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
-  // Unrolling the first digit makes a small difference on some implementations (e.g. westmere)
-  if (parse_digit(*p, i)) { ++p; }
   while (parse_digit(*p, i)) { p++; }
+}
+
+simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
+  // we continue with the fiction that we have an integer. If the
+  // floating point number is representable as x * 10^z for some integer
+  // z that fits in 53 bits, then we will be able to convert back the
+  // the integer into a float in a lossless manner.
+  const uint8_t *const first_after_period = p;
+  parse_fraction_digits(p, i);
   exponent = first_after_period - p;
   // Decimal without digits (123.) is illegal
   if (exponent == 0) {
@@ -26132,7 +30448,7 @@ simdjson_inline size_t significant_digits(const uint8_t * start_digits, size_t d
 } // unnamed namespace

 /** @private */
-static error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
+inline error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
   if (parse_float_fallback(src, answer)) {
     return SUCCESS;
   }
@@ -26215,15 +30531,17 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
   return SUCCESS;              // always succeeds
 }

-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept  { return false; }
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept  { return false; }
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept { return number_type::signed_integer; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline bool is_negative(const uint8_t *) noexcept  { return false; }
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t *) noexcept  { return false; }
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t *) noexcept { return number_type::signed_integer; }
 #else

 // parse the number at src
@@ -26254,7 +30572,21 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
   size_t digit_count = size_t(p - start_digits);
-  if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) { return INVALID_NUMBER(src); }
+  if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // By this point, we know that our input does not begin with a digit. We will attempt
+    // to handle NaN/Infinity.
+
+    double d;
+    if (compute_nan_inf(p, negative, d)) {
+      writer.append_double(d);
+      return SUCCESS;
+    }
+#endif
+
+    return INVALID_NUMBER(src);
+  }

   //
   // Handle floats if there is a . or e (or both)
@@ -26303,7 +30635,7 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
     // - Therefore, if the number is positive and lower than that, it's overflow.
     // - The value we are looking at is less than or equal to INT64_MAX.
     //
-    }  else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return INVALID_NUMBER(src); }
+    }  else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return BIGINT_NUMBER(src); }
   }

   // Write unsigned if it does not fit in a signed integer.
@@ -26402,7 +30734,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const u
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -26500,7 +30832,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_stri
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -26555,7 +30887,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uin
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -26641,7 +30973,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = src;
   uint64_t i = 0;
-  while (parse_digit(*src, i)) { src++; }
+  parse_integer_digits(src, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -26681,11 +31013,20 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   //
   uint64_t i = 0;
   const uint8_t *p = src;
-  p += parse_digit(*p, i);
-  bool leading_zero = (i == 0);
-  while (parse_digit(*p, i)) { p++; }
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
   // no integer digits, or 0123 (zero must be solo)
-  if ( p == src ) { return INCORRECT_TYPE; }
+  if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no loading digits, the number may be nan or infinity.
+    // Attempt to compute those, and return on success.
+    double d;
+    if (compute_nan_inf(p, negative, d)) { return d; }
+#endif
+
+    return INCORRECT_TYPE;
+  }
   if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }

   //
@@ -26696,9 +31037,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   if (simdjson_likely(*p == '.')) {
     p++;
     const uint8_t *start_decimal_digits = p;
-    if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
-    p++;
-    while (parse_digit(*p, i)) { p++; }
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
     exponent = -(p - start_decimal_digits);

     // Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -26747,67 +31087,12 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   return d;
 }

-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
-  return (*src == '-');
-}
-
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept {
-  bool negative = (*src == '-');
-  src += uint8_t(negative);
-  const uint8_t *p = src;
-  while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
-  if ( p == src ) { return NUMBER_ERROR; }
-  if (jsoncharutils::is_structural_or_whitespace(*p)) { return true; }
-  return false;
-}
-
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept {
-  bool negative = (*src == '-');
-  src += uint8_t(negative);
-  const uint8_t *p = src;
-  while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
-  size_t digit_count = size_t(p - src);
-  if ( p == src ) { return NUMBER_ERROR; }
-  if (jsoncharutils::is_structural_or_whitespace(*p)) {
-    static const uint8_t * smaller_big_integer = reinterpret_cast<const uint8_t *>("9223372036854775808");
-    // We have an integer.
-    if(simdjson_unlikely(digit_count > 20)) {
-      return number_type::big_integer;
-    }
-    // If the number is negative and valid, it must be a signed integer.
-    if(negative) {
-      if (simdjson_unlikely(digit_count > 19)) return number_type::big_integer;
-      if (simdjson_unlikely(digit_count == 19 && memcmp(src, smaller_big_integer, 19) > 0)) {
-        return number_type::big_integer;
-      }
-#if SIMDJSON_MINUS_ZERO_AS_FLOAT
-      if(digit_count == 1 && src[0] == '0') {
-        // We have to write -0.0 instead of 0
-        return number_type::floating_point_number;
-      }
-#endif
-      return number_type::signed_integer;
-    }
-    // Let us check if we have a big integer (>=2**64).
-    static const uint8_t * two_to_sixtyfour = reinterpret_cast<const uint8_t *>("18446744073709551616");
-    if((digit_count > 20) || (digit_count == 20 && memcmp(src, two_to_sixtyfour, 20) >= 0)) {
-      return number_type::big_integer;
-    }
-    // The number is positive and smaller than 18446744073709551616 (or 2**64).
-    // We want values larger or equal to 9223372036854775808 to be unsigned
-    // integers, and the other values to be signed integers.
-    if((digit_count == 20) || (digit_count >= 19 && memcmp(src, smaller_big_integer, 19) >= 0)) {
-      return number_type::unsigned_integer;
-    }
-    return number_type::signed_integer;
-  }
-  // Hopefully, we have 'e' or 'E' or '.'.
-  return number_type::floating_point_number;
-}
-
-// Never read at src_end or beyond
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * src, const uint8_t * const src_end) noexcept {
-  if(src == src_end) { return NUMBER_ERROR; }
+// Parse a JSON number into a binary32 (float) value.
+//
+// This mirrors parse_double, but it rounds to binary32 directly instead of
+// rounding to binary64 and then to binary32: the latter is subject to double
+// rounding and would not always produce the nearest float.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * src) noexcept {
   //
   // Check for minus sign
   //
@@ -26819,91 +31104,20 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   //
   uint64_t i = 0;
   const uint8_t *p = src;
-  if(p == src_end) { return NUMBER_ERROR; }
-  p += parse_digit(*p, i);
-  bool leading_zero = (i == 0);
-  while ((p != src_end) && parse_digit(*p, i)) { p++; }
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
   // no integer digits, or 0123 (zero must be solo)
-  if ( p == src ) { return INCORRECT_TYPE; }
-  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+  if ( p == src ) {

-  //
-  // Parse the decimal part.
-  //
-  int64_t exponent = 0;
-  bool overflow;
-  if (simdjson_likely((p != src_end) && (*p == '.'))) {
-    p++;
-    const uint8_t *start_decimal_digits = p;
-    if ((p == src_end) || !parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
-    p++;
-    while ((p != src_end) && parse_digit(*p, i)) { p++; }
-    exponent = -(p - start_decimal_digits);
-
-    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
-    overflow = p-src-1 > 19;
-    if (simdjson_unlikely(overflow && leading_zero)) {
-      // Skip leading 0.00000 and see if it still overflows
-      const uint8_t *start_digits = src + 2;
-      while (*start_digits == '0') { start_digits++; }
-      overflow = start_digits-src > 19;
-    }
-  } else {
-    overflow = p-src > 19;
-  }
-
-  //
-  // Parse the exponent
-  //
-  if ((p != src_end) && (*p == 'e' || *p == 'E')) {
-    p++;
-    if(p == src_end) { return NUMBER_ERROR; }
-    bool exp_neg = *p == '-';
-    p += exp_neg || *p == '+';
-
-    uint64_t exp = 0;
-    const uint8_t *start_exp_digits = p;
-    while ((p != src_end) && parse_digit(*p, exp)) { p++; }
-    // no exp digits, or 20+ exp digits
-    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
-
-    exponent += exp_neg ? 0-exp : exp;
-  }
-
-  if ((p != src_end) && jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
-
-  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, the number may be nan or infinity.
+    // Attempt to compute those, and return on success.
+    float f;
+    if (compute_nan_inf(p, negative, f)) { return f; }
+#endif

-  //
-  // Assemble (or slow-parse) the float
-  //
-  double d;
-  if (simdjson_likely(!overflow)) {
-    if (compute_float_64(exponent, i, negative, d)) { return d; }
-  }
-  if (!parse_float_fallback(src - uint8_t(negative), src_end, &d)) {
-    return NUMBER_ERROR;
+    return INCORRECT_TYPE;
   }
-  return d;
-}
-
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * src) noexcept {
-  //
-  // Check for minus sign
-  //
-  bool negative = (*(src + 1) == '-');
-  src += uint8_t(negative) + 1;
-
-  //
-  // Parse the integer part.
-  //
-  uint64_t i = 0;
-  const uint8_t *p = src;
-  p += parse_digit(*p, i);
-  bool leading_zero = (i == 0);
-  while (parse_digit(*p, i)) { p++; }
-  // no integer digits, or 0123 (zero must be solo)
-  if ( p == src ) { return INCORRECT_TYPE; }
   if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }

   //
@@ -26914,9 +31128,239 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   if (simdjson_likely(*p == '.')) {
     p++;
     const uint8_t *start_decimal_digits = p;
-    if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
-    p++;
-    while (parse_digit(*p, i)) { p++; }
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = p-start_digits > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if (*p == 'e' || *p == 'E') {
+    p++;
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while (parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  float d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_32(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
+simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
+  return (*src == '-');
+}
+
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept {
+  bool negative = (*src == '-');
+  src += uint8_t(negative);
+  const uint8_t *p = src;
+  while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
+  if ( p == src ) { return NUMBER_ERROR; }
+  if (jsoncharutils::is_structural_or_whitespace(*p)) { return true; }
+  return false;
+}
+
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept {
+  bool negative = (*src == '-');
+  src += uint8_t(negative);
+  const uint8_t *p = src;
+  while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
+  size_t digit_count = size_t(p - src);
+  if ( p == src ) { return NUMBER_ERROR; }
+  if (jsoncharutils::is_structural_or_whitespace(*p)) {
+    static const uint8_t * smaller_big_integer = reinterpret_cast<const uint8_t *>("9223372036854775808");
+    // We have an integer.
+    if(simdjson_unlikely(digit_count > 20)) {
+      return number_type::big_integer;
+    }
+    // If the number is negative and valid, it must be a signed integer.
+    if(negative) {
+      if (simdjson_unlikely(digit_count > 19)) return number_type::big_integer;
+      if (simdjson_unlikely(digit_count == 19 && memcmp(src, smaller_big_integer, 19) > 0)) {
+        return number_type::big_integer;
+      }
+#if SIMDJSON_MINUS_ZERO_AS_FLOAT
+      if(digit_count == 1 && src[0] == '0') {
+        // We have to write -0.0 instead of 0
+        return number_type::floating_point_number;
+      }
+#endif
+      return number_type::signed_integer;
+    }
+    // Let us check if we have a big integer (>=2**64).
+    static const uint8_t * two_to_sixtyfour = reinterpret_cast<const uint8_t *>("18446744073709551616");
+    if((digit_count > 20) || (digit_count == 20 && memcmp(src, two_to_sixtyfour, 20) >= 0)) {
+      return number_type::big_integer;
+    }
+    // The number is positive and smaller than 18446744073709551616 (or 2**64).
+    // We want values larger or equal to 9223372036854775808 to be unsigned
+    // integers, and the other values to be signed integers.
+    if((digit_count == 20) || (digit_count >= 19 && memcmp(src, smaller_big_integer, 19) >= 0)) {
+      return number_type::unsigned_integer;
+    }
+    return number_type::signed_integer;
+  }
+  // Hopefully, we have 'e' or 'E' or '.'.
+  return number_type::floating_point_number;
+}
+
+// Never read at src_end or beyond
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * src, const uint8_t * const src_end) noexcept {
+  if(src == src_end) { return NUMBER_ERROR; }
+  //
+  // Check for minus sign
+  //
+  bool negative = (*src == '-');
+  src += uint8_t(negative);
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  if(p == src_end) { return NUMBER_ERROR; }
+  p += parse_digit(*p, i);
+  bool leading_zero = (i == 0);
+  while ((p != src_end) && parse_digit(*p, i)) { p++; }
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) { return INCORRECT_TYPE; }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely((p != src_end) && (*p == '.'))) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    if ((p == src_end) || !parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
+    p++;
+    while ((p != src_end) && parse_digit(*p, i)) { p++; }
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = start_digits-src > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if ((p != src_end) && (*p == 'e' || *p == 'E')) {
+    p++;
+    if(p == src_end) { return NUMBER_ERROR; }
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while ((p != src_end) && parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if ((p != src_end) && jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  double d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_64(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), src_end, &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * src) noexcept {
+  //
+  // Check for minus sign
+  //
+  bool negative = (*(src + 1) == '-');
+  src += uint8_t(negative) + 1;
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, attempt to parse numbers that are either
+    // NaN or Infinity
+    if (atomparsing::is_valid_inf_in_string(src)) {
+      double inf = std::numeric_limits<double>::infinity();
+      return negative ? -inf : inf;
+    }
+
+    if (atomparsing::is_valid_nan_in_string(src)) {
+      return std::numeric_limits<double>::quiet_NaN();
+    }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
     exponent = -(p - start_decimal_digits);

     // Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -26965,6 +31409,99 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   return d;
 }

+// Parse a JSON number held inside a JSON string into a binary32 (float) value.
+// See parse_float for why we do not simply round parse_double_in_string.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * src) noexcept {
+  //
+  // Check for minus sign
+  //
+  bool negative = (*(src + 1) == '-');
+  src += uint8_t(negative) + 1;
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, attempt to parse numbers that are either
+    // NaN or Infinity
+    if (atomparsing::is_valid_inf_in_string(src)) {
+      float inf = std::numeric_limits<float>::infinity();
+      return negative ? -inf : inf;
+    }
+
+    if (atomparsing::is_valid_nan_in_string(src)) {
+      return std::numeric_limits<float>::quiet_NaN();
+    }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = p-start_digits > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if (*p == 'e' || *p == 'E') {
+    p++;
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while (parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if (*p != '"') { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  float d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_32(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
 } // unnamed namespace
 #endif // SIMDJSON_SKIPNUMBERPARSING

@@ -27269,16 +31806,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
 }
 #endif

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
-                                uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
-  return _addcarry_u64(0, value1, value2,
-                       reinterpret_cast<unsigned __int64 *>(result));
-#else
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-#endif
-}

 } // unnamed namespace
 } // namespace westmere
@@ -27857,16 +32384,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
 }
 #endif

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
-                                uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
-  return _addcarry_u64(0, value1, value2,
-                       reinterpret_cast<unsigned __int64 *>(result));
-#else
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-#endif
-}

 } // unnamed namespace
 } // namespace westmere
@@ -28480,6 +32997,9 @@ namespace atomparsing {
 // to the compile-time constant 1936482662.
 simdjson_inline uint32_t string_to_uint32(const char* str) { uint32_t val; std::memcpy(&val, str, sizeof(uint32_t)); return val; }

+// Acts on the same principle as string_to_uint32, but on an 8-byte block of memory
+simdjson_inline uint64_t string_to_uint64(const char* str) { uint64_t val; std::memcpy(&val, str, sizeof(uint64_t)); return val; }
+

 // Again in str4ncmp we use a memcpy to avoid undefined behavior. The memcpy may appear expensive.
 // Yet all decent optimizing compilers will compile memcpy to a single instruction, just about.
@@ -28491,6 +33011,28 @@ simdjson_inline uint32_t str4ncmp(const uint8_t *src, const char* atom) {
   return srcval ^ string_to_uint32(atom);
 }

+// Checks that the first 8 characters of the input string match the given atom in a case-insensitive manner.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint64_t str8ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+  uint64_t srcval; // we want to avoid unaligned 32-bit loads (undefined in C/C++)
+  static_assert(sizeof(uint64_t) <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be larger than 8 bytes");
+  std::memcpy(&srcval, src, sizeof(uint64_t));
+
+  return (srcval | 0x2020202020202020ull) ^ string_to_uint64(atom);
+}
+
+// Checks that the first 3 characters of 'src' match 'atom' in a case-insensitive way.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint32_t str3ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+  return ((src[0] | 0x20) ^ atom[0]) //
+       | ((src[1] | 0x20) ^ atom[1]) //
+       | ((src[2] | 0x20) ^ atom[2]);
+}
+
 simdjson_warn_unused
 simdjson_inline bool is_valid_true_atom(const uint8_t *src) {
   return (str4ncmp(src, "true") | jsoncharutils::is_not_structural_or_whitespace(src[4])) == 0;
@@ -28527,6 +33069,71 @@ simdjson_inline bool is_valid_null_atom(const uint8_t *src, size_t len) {
   else { return false; }
 }

+#if SIMDJSON_ENABLE_NAN_INF
+// "nan" is 3 bytes; we check characters and then verify the next
+// character is structural or whitespace. We accept both "nan" and "NaN".
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src) {
+  return (str3ncmp_case_insensitive(src, "nan")
+        | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+}
+
+// checks that the next four characters of a string are 'nan"', where the 'nan'
+// is checked in a case-insensitive way.
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_in_string(const uint8_t *src) {
+  return (str3ncmp_case_insensitive(src, "nan") | (src[3] ^ '"')) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src, size_t len) {
+  if (len > 3) { return is_valid_nan_atom(src); }
+  if (len == 3) { return str3ncmp_case_insensitive(src, "nan") == 0; }
+  return false;
+}
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src) {
+  bool is_short_inf = (str3ncmp_case_insensitive(src, "inf")
+                     | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+  if(is_short_inf) return true;
+
+  // Check for 'infinity' (any capitalization)
+  return (str8ncmp_case_insensitive(src, "infinity") | jsoncharutils::is_not_structural_or_whitespace(src[8])) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_in_string(const uint8_t *src) {
+  bool is_short_inf = (str3ncmp_case_insensitive(src, "inf") | (src[3] ^ '"')) == 0;
+  if(is_short_inf) return true;
+
+  return (str8ncmp_case_insensitive(src, "infinity") | (src[8] ^ '"')) == 0;
+}
+
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src, size_t len) {
+  if (len > 8) { return is_valid_inf_atom(src); }
+  if (len == 8 && str8ncmp_case_insensitive(src, "infinity") == 0) {
+    return true;
+  }
+  if (len > 3) {
+    return (str3ncmp_case_insensitive(src, "inf")
+          | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+  }
+  if (len == 3) { return str3ncmp_case_insensitive(src, "inf") == 0; }
+  return false;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
 } // namespace atomparsing
 } // unnamed namespace
 } // namespace westmere
@@ -28617,7 +33224,7 @@ inline simdjson_warn_unused error_code dom_parser_implementation::set_capacity(s
 }

 inline simdjson_warn_unused error_code dom_parser_implementation::set_max_depth(size_t max_depth) noexcept {
-  if(max_depth == 0 || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
+  if(max_depth == 0 || max_depth > SIMDJSON_MAX_DEPTH || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
   // Stage 2 stacks
   open_containers.reset(new (std::nothrow) open_container[max_depth]);
   is_array.reset(new (std::nothrow) bool[max_depth]);
@@ -28796,6 +33403,7 @@ protected:
 /* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_NUMBERPARSING_H */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/base.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/jsoncharutils.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/atomparsing.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/internal/numberparsing_tables.h" */
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */

@@ -28835,6 +33443,18 @@ simdjson_inline double to_double(uint64_t mantissa, uint64_t real_exponent, bool
     return d;
 }

+// Convert a mantissa, an exponent and a sign bit into an ieee32 float (binary32).
+// The real_exponent needs to be in [0, 254] (technically real_exponent = 255 would be acceptable).
+// The mantissa should be in [0,1<<24). The bit at index (1U << 23) will be zeroed.
+simdjson_inline float to_float(uint32_t mantissa, uint32_t real_exponent, bool negative) {
+    float f;
+    mantissa &= ~(uint32_t(1) << 23);
+    mantissa |= real_exponent << 23;
+    mantissa |= ((static_cast<uint32_t>(negative)) << 31);
+    std::memcpy(&f, &mantissa, sizeof(f));
+    return f;
+}
+
 // Attempts to compute i * 10^(power) exactly; and if "negative" is
 // true, negate the result.
 // This function will only work in some cases, when it does not work, success is
@@ -29091,6 +33711,219 @@ simdjson_inline bool compute_float_64(int64_t power, uint64_t i, bool negative,
   return true;
 }

+// Attempts to compute i * 10^(power) as a binary32 (float) value; and if
+// "negative" is true, negate the result.
+//
+// This is the same Eisel-Lemire algorithm as compute_float_64, retargeted at
+// binary32. It is adapted from fast_float
+// (https://github.com/fastfloat/fast_float), where the routine is written once
+// and instantiated for each binary format. We only need 24 bits of mantissa
+// instead of 53, so we ask the truncated product for 24 + 2 = 26 accurate bits
+// (plus one bit that may be lost to the "upperbit" shift), and we shift the
+// product right by (upperbit + 64 - 23 - 3) instead of (upperbit + 64 - 52 - 3).
+//
+// The power-of-five table is shared with the binary64 code: the useful range
+// for binary32, [smallest_power_binary32, largest_power_binary32], is a strict
+// subset of [smallest_power, largest_power].
+//
+// The function returns false when the result would be infinite: simdjson
+// refuses to parse infinite values, so the caller reports an error. Unlike
+// compute_float_64, a false return never means "try harder": the accuracy of
+// the two-limb product is guaranteed by Noble Mushtak and Daniel Lemire,
+// "Fast Number Parsing Without Fallback" (https://arxiv.org/abs/2212.06644).
+// The caller must still fall back to a slow path when the 64-bit mantissa i
+// was truncated (more than 19 significant digits).
+//
+// We assume that power is in the [smallest_power, largest_power] interval: the
+// caller is responsible for this check.
+simdjson_inline bool compute_float_32(int64_t power, uint64_t i, bool negative, float &d) {
+  // Powers of ten that are exactly representable as binary32 values: 10^k is
+  // exact as long as k <= 10 (5^10 = 9765625 < 2**24 while 5^11 > 2**24).
+  static constexpr float power_of_ten_float[] = {
+      1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+      1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+  // The range of powers of ten that a non-zero, finite binary32 value can be
+  // built from. Anything smaller rounds to zero, anything larger is infinite.
+  // These are the smallest_power_of_ten()/largest_power_of_ten() constants that
+  // fast_float uses for binary32.
+  constexpr int smallest_power_binary32 = -65;
+  constexpr int largest_power_binary32 = 38;
+
+  // We start with the fast path described in
+  // Clinger WD. How to read floating point numbers accurately.
+  // ACM SIGPLAN Notices. 1990
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+  // We cannot be certain that x/y is rounded to nearest.
+  if (0 <= power && power <= 10 && i <= 16777215)
+#else
+  if (-10 <= power && power <= 10 && i <= 16777215)
+#endif
+  {
+    // Convert the integer into a float. This is lossless since
+    // 0 <= i <= 2^24 - 1.
+    d = float(i);
+    // Both d and the power of ten are exactly representable as binary32
+    // values, so the product (or quotient) is correctly rounded.
+    if (power < 0) {
+      d = d / power_of_ten_float[-power];
+    } else {
+      d = d * power_of_ten_float[power];
+    }
+    if (negative) {
+      d = -d;
+    }
+    return true;
+  }
+
+  // The fast path has failed, so we fall back on the Eisel-Lemire algorithm.
+  // It needs i > 0 (so that the leading bit of i can be normalized), so we
+  // handle i == 0 separately. We also handle the powers of ten that are so
+  // small (or so large) that the answer is zero (or infinite) whatever the
+  // mantissa is.
+  if (i == 0 || power < smallest_power_binary32) {
+    d = negative ? -0.0f : 0.0f;
+    return true;
+  }
+  if (power > largest_power_binary32) {
+    // We have, for sure, an infinite value.
+    return false;
+  }
+
+  // The exponent is 128 + 63 + power + floor(log(5**power)/log(2)).
+  // The 128 comes from the ieee32 standard (the minimal exponent is -127).
+  // The 63 comes from the fact that we use a 64-bit word.
+  // See compute_float_64 for a discussion of the magical 152170 + 65536.
+  int64_t exponent = (((152170 + 65536) * power) >> 16) + 128 + 63;
+
+  // We want the most significant bit of i to be 1. Shift if needed.
+  int lz = leading_zeroes(i);
+  i <<= lz;
+
+  // We want the most significant 64 bits of the product i * 5**power. It is
+  // safe to index the table because
+  // smallest_power <= smallest_power_binary32 <= power
+  //               <= largest_power_binary32 <= largest_power.
+  const uint32_t index = 2 * uint32_t(power - simdjson::internal::smallest_power);
+#if SIMDJSON_STATIC_REFLECTION
+  simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index]);
+#else
+  simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index]);
+#endif
+
+  // Unless the least significant 38 bits of the high (64-bit) part of the full
+  // product are all 1s, then we know that the most significant 26 bits are
+  // exact and no further work is needed. Having 26 bits is necessary because
+  // we need 24 bits for the mantissa but we have to have one rounding bit and
+  // we can waste a bit if the most significant bit of the product is zero.
+  // (For binary64, the same reasoning gives 55 bits and a 9-bit mask.)
+  if((firstproduct.high & 0x3FFFFFFFFF) == 0x3FFFFFFFFF) {
+    // The truncated multiplication was not accurate enough; use the next 64
+    // bits of the power of five to refine it. See compute_float_64 for a
+    // detailed discussion.
+#if SIMDJSON_STATIC_REFLECTION
+    simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index + 1]);
+#else
+    simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index + 1]);
+#endif
+    firstproduct.low += secondproduct.high;
+    if(secondproduct.high > firstproduct.low) { firstproduct.high++; }
+  }
+  uint64_t lower = firstproduct.low;
+  uint64_t upper = firstproduct.high;
+  // The final mantissa should be 24 bits with a leading 1.
+  // We shift it so that it occupies 25 bits with a leading 1.
+  ///////
+  uint64_t upperbit = upper >> 63;
+  uint64_t mantissa = upper >> (upperbit + 38); // 38 == 64 - 23 - 3
+  lz += int(1 ^ upperbit);
+
+  // Here we have mantissa < (1<<25).
+  int64_t real_exponent = exponent - lz;
+  if (simdjson_unlikely(real_exponent <= 0)) { // we have a subnormal?
+    // Here we have that real_exponent <= 0 so -real_exponent >= 0
+    if(-real_exponent + 1 >= 64) { // if we have more than 64 bits below the minimum exponent, you have a zero for sure.
+      d = negative ? -0.0f : 0.0f;
+      return true;
+    }
+    // next line is safe because -real_exponent + 1 < 64
+    mantissa >>= -real_exponent + 1;
+    // Thankfully, we can't have both "round-to-even" and subnormals because
+    // "round-to-even" only occurs for powers close to 0.
+    mantissa += (mantissa & 1); // round up
+    mantissa >>= 1;
+    // As in compute_float_64, rounding up may take us out of the subnormal
+    // range, so we can only decide after rounding.
+    real_exponent = (mantissa < (uint64_t(1) << 23)) ? 0 : 1;
+    d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+    return true;
+  }
+  // We have to round to even. The "to even" part is only a problem when we are
+  // right in between two floats, which we guard against. The bounds on the
+  // power of ten are those of fast_float's
+  // min_exponent_round_to_even()/max_exponent_round_to_even() for binary32:
+  // when q >= 0, (2m+1) must be divisible by 5^q with 5^q <= 2^25, so q <= 10;
+  // when q < 0, we need 2^24 x 5^{-q} < 2^{64}, so q >= -17.
+  if (simdjson_unlikely((lower <= 1) && (power >= -17) && (power <= 10) && ((mantissa & 3) == 1))) {
+    if((mantissa << (upperbit + 38)) == upper) {
+      mantissa &= ~uint64_t(1);   // flip it so that we do not round up
+    }
+  }
+
+  mantissa += mantissa & 1;
+  mantissa >>= 1;
+
+  // Here we have mantissa < (1<<24), unless there was an overflow
+  if (mantissa >= (uint64_t(1) << 24)) {
+    mantissa = (uint64_t(1) << 23);
+    real_exponent++;
+  }
+  mantissa &= ~(uint64_t(1) << 23);
+  // we have to check that real_exponent is in range, otherwise we bail out
+  if (simdjson_unlikely(real_exponent > 254)) {
+    // We have an infinite value!!! We could actually throw an error here if we could.
+    return false;
+  }
+  d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+  return true;
+}
+
+#if SIMDJSON_ENABLE_NAN_INF
+// Parses a nan or infinity. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, double& d) noexcept {
+  if (atomparsing::is_valid_inf_atom(src)) {
+    double inf = std::numeric_limits<double>::infinity();
+    d = negative ? -inf : inf;
+    return true;
+  }
+
+  if (atomparsing::is_valid_nan_atom(src)) {
+    d = std::numeric_limits<double>::quiet_NaN();
+    return true;
+  }
+
+  return false;
+}
+
+// Parses a nan or infinity as a binary32 value. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, float& d) noexcept {
+  if (atomparsing::is_valid_inf_atom(src)) {
+    float inf = std::numeric_limits<float>::infinity();
+    d = negative ? -inf : inf;
+    return true;
+  }
+
+  if (atomparsing::is_valid_nan_atom(src)) {
+    d = std::numeric_limits<float>::quiet_NaN();
+    return true;
+  }
+
+  return false;
+}
+#endif
+
 // We call a fallback floating-point parser that might be slow. Note
 // it will accept JSON numbers, but the JSON spec. is more restrictive so
 // before you call parse_float_fallback, you need to have validated the input
@@ -29128,6 +33961,16 @@ static bool parse_float_fallback(const uint8_t *ptr, const uint8_t *end_ptr, dou
   return !(*outDouble > (std::numeric_limits<double>::max)() || *outDouble < std::numeric_limits<double>::lowest());
 }

+// Same as parse_float_fallback, but for binary32 (float) values. Going through
+// the binary64 fallback and then rounding to binary32 would be subject to
+// double rounding, so we run the fallback algorithm directly on binary32.
+static bool parse_float_fallback(const uint8_t *ptr, float *outFloat) {
+  *outFloat = simdjson::internal::from_chars_float(reinterpret_cast<const char *>(ptr));
+  // We do not accept infinite values. See the binary64 version above for why we
+  // do not use std::isfinite.
+  return !(*outFloat > (std::numeric_limits<float>::max)() || *outFloat < std::numeric_limits<float>::lowest());
+}
+
 // check quickly whether the next 8 chars are made of digits
 // at a glance, it looks better than Mula's
 // http://0x80.pl/articles/swar-digits-validate.html
@@ -29146,6 +33989,28 @@ simdjson_inline bool is_made_of_eight_digits_fast(const uint8_t *chars) {
           0x3333333333333333);
 }

+// The same idea for four characters. A block of eight only bites when eight
+// digits are left, so a run of 4 to 7 digits used to fall back to one digit at a
+// time; taking four of them at once is what makes the tail of a long fraction
+// cheap. Adapted from fast_float, credit @aqrit.
+simdjson_inline bool is_made_of_four_digits_fast(const uint8_t *chars) {
+  uint32_t val;
+  // Reads up to 3 bytes beyond the digits, which SIMDJSON_PADDING covers.
+  static_assert(3 <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be bigger than 3");
+  std::memcpy(&val, chars, 4);
+  return !(((val + 0x46464646) | (val - 0x30303030)) & 0x80808080);
+}
+
+// Only call this when is_made_of_four_digits_fast() says the four characters are
+// digits. Little-endian: guarded by SIMDJSON_SWAR_NUMBER_PARSING.
+simdjson_inline uint32_t parse_four_digits_unrolled(const uint8_t *chars) {
+  uint32_t val;
+  std::memcpy(&val, chars, 4);
+  val -= 0x30303030;
+  val = (val * 10) + (val >> 8);
+  return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
 template<typename I>
 SIMDJSON_NO_SANITIZE_UNDEFINED // We deliberately allow overflow here and check later
 simdjson_inline bool parse_digit(const uint8_t c, I &i) {
@@ -29162,26 +34027,73 @@ simdjson_inline bool is_digit(const uint8_t c) {
   return static_cast<uint8_t>(c - '0') <= 9;
 }

-simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
-  // we continue with the fiction that we have an integer. If the
-  // floating point number is representable as x * 10^z for some integer
-  // z that fits in 53 bits, then we will be able to convert back the
-  // the integer into a float in a lossless manner.
-  const uint8_t *const first_after_period = p;
+// Consumes a run of digits into i, eight at a time while eight are available,
+// then four, then one at a time. Overflow is deliberate: the caller counts the
+// digits and falls back when there are too many for a 64-bit mantissa.
+//
+// Long runs of digits are what a float-heavy document is made of: the fraction
+// of a binary64 printed to full precision is 15 to 17 digits, and reading those
+// one at a time is the single largest cost in parse_double(). Integer parts are
+// left alone: they are usually a handful of digits, and there the check for a
+// block of eight is wasted work.
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_fraction_digits(const uint8_t *&p, uint64_t &i) {
+#ifdef SIMDJSON_SWAR_NUMBER_PARSING
+#if SIMDJSON_SWAR_NUMBER_PARSING
+  while (is_made_of_eight_digits_fast(p)) {
+    i = i * 100000000 + parse_eight_digits_unrolled(p);
+    p += 8;
+  }
+  // A 4 to 7 digit remainder is the common case once the blocks of eight are
+  // gone: 15 fraction digits leave 7, and 17 leave 1 after two blocks.
+  if (is_made_of_four_digits_fast(p)) {
+    i = i * 10000 + parse_four_digits_unrolled(p);
+    p += 4;
+  }
+#endif // SIMDJSON_SWAR_NUMBER_PARSING
+#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
+  while (parse_digit(*p, i)) { p++; }
+}

+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_float_integer_digits(const uint8_t *&p, uint64_t &i,
+                                                bool &leading_zero) {
+  if (!parse_digit(*p, i)) { leading_zero = true; return; }
+  p++;
+  leading_zero = (i == 0);
+  while (parse_digit(*p, i)) { p++; }
+}
+
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_integer_digits(const uint8_t *&p, uint64_t &i) {
 #ifdef SIMDJSON_SWAR_NUMBER_PARSING
 #if SIMDJSON_SWAR_NUMBER_PARSING
-  // this helps if we have lots of decimals!
-  // this turns out to be frequent enough.
+  // Identifiers, timestamps and counters often have eight digits or more.
+  // Prior related work: jsonifier parses integers as eight-digit SWAR words
+  // (str_to_i.hpp, https://github.com/nihilai-collective/Jsonifier). This
+  // takes one such word with parse_eight_digits_unrolled, the routine
+  // simdjson uses for long fractions.
   if (is_made_of_eight_digits_fast(p)) {
     i = i * 100000000 + parse_eight_digits_unrolled(p);
     p += 8;
   }
+  const uint8_t *const swar_end = p + 8;
+  while (p < swar_end && is_made_of_four_digits_fast(p)) {
+    i = i * 10000 + parse_four_digits_unrolled(p);
+    p += 4;
+  }
 #endif // SIMDJSON_SWAR_NUMBER_PARSING
 #endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
-  // Unrolling the first digit makes a small difference on some implementations (e.g. westmere)
-  if (parse_digit(*p, i)) { ++p; }
   while (parse_digit(*p, i)) { p++; }
+}
+
+simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
+  // we continue with the fiction that we have an integer. If the
+  // floating point number is representable as x * 10^z for some integer
+  // z that fits in 53 bits, then we will be able to convert back the
+  // the integer into a float in a lossless manner.
+  const uint8_t *const first_after_period = p;
+  parse_fraction_digits(p, i);
   exponent = first_after_period - p;
   // Decimal without digits (123.) is illegal
   if (exponent == 0) {
@@ -29270,7 +34182,7 @@ simdjson_inline size_t significant_digits(const uint8_t * start_digits, size_t d
 } // unnamed namespace

 /** @private */
-static error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
+inline error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
   if (parse_float_fallback(src, answer)) {
     return SUCCESS;
   }
@@ -29353,15 +34265,17 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
   return SUCCESS;              // always succeeds
 }

-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept  { return false; }
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept  { return false; }
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept { return number_type::signed_integer; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline bool is_negative(const uint8_t *) noexcept  { return false; }
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t *) noexcept  { return false; }
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t *) noexcept { return number_type::signed_integer; }
 #else

 // parse the number at src
@@ -29392,7 +34306,21 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
   size_t digit_count = size_t(p - start_digits);
-  if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) { return INVALID_NUMBER(src); }
+  if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // By this point, we know that our input does not begin with a digit. We will attempt
+    // to handle NaN/Infinity.
+
+    double d;
+    if (compute_nan_inf(p, negative, d)) {
+      writer.append_double(d);
+      return SUCCESS;
+    }
+#endif
+
+    return INVALID_NUMBER(src);
+  }

   //
   // Handle floats if there is a . or e (or both)
@@ -29441,7 +34369,7 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
     // - Therefore, if the number is positive and lower than that, it's overflow.
     // - The value we are looking at is less than or equal to INT64_MAX.
     //
-    }  else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return INVALID_NUMBER(src); }
+    }  else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return BIGINT_NUMBER(src); }
   }

   // Write unsigned if it does not fit in a signed integer.
@@ -29540,7 +34468,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const u
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -29638,7 +34566,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_stri
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -29693,7 +34621,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uin
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -29779,7 +34707,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = src;
   uint64_t i = 0;
-  while (parse_digit(*src, i)) { src++; }
+  parse_integer_digits(src, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -29819,11 +34747,20 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   //
   uint64_t i = 0;
   const uint8_t *p = src;
-  p += parse_digit(*p, i);
-  bool leading_zero = (i == 0);
-  while (parse_digit(*p, i)) { p++; }
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
   // no integer digits, or 0123 (zero must be solo)
-  if ( p == src ) { return INCORRECT_TYPE; }
+  if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no loading digits, the number may be nan or infinity.
+    // Attempt to compute those, and return on success.
+    double d;
+    if (compute_nan_inf(p, negative, d)) { return d; }
+#endif
+
+    return INCORRECT_TYPE;
+  }
   if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }

   //
@@ -29834,9 +34771,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   if (simdjson_likely(*p == '.')) {
     p++;
     const uint8_t *start_decimal_digits = p;
-    if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
-    p++;
-    while (parse_digit(*p, i)) { p++; }
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
     exponent = -(p - start_decimal_digits);

     // Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -29885,6 +34821,97 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   return d;
 }

+// Parse a JSON number into a binary32 (float) value.
+//
+// This mirrors parse_double, but it rounds to binary32 directly instead of
+// rounding to binary64 and then to binary32: the latter is subject to double
+// rounding and would not always produce the nearest float.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * src) noexcept {
+  //
+  // Check for minus sign
+  //
+  bool negative = (*src == '-');
+  src += uint8_t(negative);
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, the number may be nan or infinity.
+    // Attempt to compute those, and return on success.
+    float f;
+    if (compute_nan_inf(p, negative, f)) { return f; }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = p-start_digits > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if (*p == 'e' || *p == 'E') {
+    p++;
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while (parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  float d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_32(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
 simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
   return (*src == '-');
 }
@@ -30037,11 +35064,25 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   //
   uint64_t i = 0;
   const uint8_t *p = src;
-  p += parse_digit(*p, i);
-  bool leading_zero = (i == 0);
-  while (parse_digit(*p, i)) { p++; }
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
   // no integer digits, or 0123 (zero must be solo)
-  if ( p == src ) { return INCORRECT_TYPE; }
+  if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, attempt to parse numbers that are either
+    // NaN or Infinity
+    if (atomparsing::is_valid_inf_in_string(src)) {
+      double inf = std::numeric_limits<double>::infinity();
+      return negative ? -inf : inf;
+    }
+
+    if (atomparsing::is_valid_nan_in_string(src)) {
+      return std::numeric_limits<double>::quiet_NaN();
+    }
+#endif
+
+    return INCORRECT_TYPE;
+  }
   if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }

   //
@@ -30052,9 +35093,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   if (simdjson_likely(*p == '.')) {
     p++;
     const uint8_t *start_decimal_digits = p;
-    if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
-    p++;
-    while (parse_digit(*p, i)) { p++; }
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
     exponent = -(p - start_decimal_digits);

     // Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -30103,6 +35143,99 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   return d;
 }

+// Parse a JSON number held inside a JSON string into a binary32 (float) value.
+// See parse_float for why we do not simply round parse_double_in_string.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * src) noexcept {
+  //
+  // Check for minus sign
+  //
+  bool negative = (*(src + 1) == '-');
+  src += uint8_t(negative) + 1;
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, attempt to parse numbers that are either
+    // NaN or Infinity
+    if (atomparsing::is_valid_inf_in_string(src)) {
+      float inf = std::numeric_limits<float>::infinity();
+      return negative ? -inf : inf;
+    }
+
+    if (atomparsing::is_valid_nan_in_string(src)) {
+      return std::numeric_limits<float>::quiet_NaN();
+    }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = p-start_digits > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if (*p == 'e' || *p == 'E') {
+    p++;
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while (parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if (*p != '"') { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  float d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_32(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
 } // unnamed namespace
 #endif // SIMDJSON_SKIPNUMBERPARSING

@@ -30368,10 +35501,6 @@ simdjson_inline int count_ones(uint64_t input_num) {
   return __lasx_xvpickve2gr_w(__lasx_xvpcnt_d(__m256i(v4u64{input_num, 0, 0, 0})), 0);
 }

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2, uint64_t *result) {
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-}

 } // unnamed namespace
 } // namespace lasx
@@ -31118,6 +36247,9 @@ namespace atomparsing {
 // to the compile-time constant 1936482662.
 simdjson_inline uint32_t string_to_uint32(const char* str) { uint32_t val; std::memcpy(&val, str, sizeof(uint32_t)); return val; }

+// Acts on the same principle as string_to_uint32, but on an 8-byte block of memory
+simdjson_inline uint64_t string_to_uint64(const char* str) { uint64_t val; std::memcpy(&val, str, sizeof(uint64_t)); return val; }
+

 // Again in str4ncmp we use a memcpy to avoid undefined behavior. The memcpy may appear expensive.
 // Yet all decent optimizing compilers will compile memcpy to a single instruction, just about.
@@ -31129,6 +36261,28 @@ simdjson_inline uint32_t str4ncmp(const uint8_t *src, const char* atom) {
   return srcval ^ string_to_uint32(atom);
 }

+// Checks that the first 8 characters of the input string match the given atom in a case-insensitive manner.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint64_t str8ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+  uint64_t srcval; // we want to avoid unaligned 32-bit loads (undefined in C/C++)
+  static_assert(sizeof(uint64_t) <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be larger than 8 bytes");
+  std::memcpy(&srcval, src, sizeof(uint64_t));
+
+  return (srcval | 0x2020202020202020ull) ^ string_to_uint64(atom);
+}
+
+// Checks that the first 3 characters of 'src' match 'atom' in a case-insensitive way.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint32_t str3ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+  return ((src[0] | 0x20) ^ atom[0]) //
+       | ((src[1] | 0x20) ^ atom[1]) //
+       | ((src[2] | 0x20) ^ atom[2]);
+}
+
 simdjson_warn_unused
 simdjson_inline bool is_valid_true_atom(const uint8_t *src) {
   return (str4ncmp(src, "true") | jsoncharutils::is_not_structural_or_whitespace(src[4])) == 0;
@@ -31165,6 +36319,71 @@ simdjson_inline bool is_valid_null_atom(const uint8_t *src, size_t len) {
   else { return false; }
 }

+#if SIMDJSON_ENABLE_NAN_INF
+// "nan" is 3 bytes; we check characters and then verify the next
+// character is structural or whitespace. We accept both "nan" and "NaN".
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src) {
+  return (str3ncmp_case_insensitive(src, "nan")
+        | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+}
+
+// checks that the next four characters of a string are 'nan"', where the 'nan'
+// is checked in a case-insensitive way.
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_in_string(const uint8_t *src) {
+  return (str3ncmp_case_insensitive(src, "nan") | (src[3] ^ '"')) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src, size_t len) {
+  if (len > 3) { return is_valid_nan_atom(src); }
+  if (len == 3) { return str3ncmp_case_insensitive(src, "nan") == 0; }
+  return false;
+}
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src) {
+  bool is_short_inf = (str3ncmp_case_insensitive(src, "inf")
+                     | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+  if(is_short_inf) return true;
+
+  // Check for 'infinity' (any capitalization)
+  return (str8ncmp_case_insensitive(src, "infinity") | jsoncharutils::is_not_structural_or_whitespace(src[8])) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_in_string(const uint8_t *src) {
+  bool is_short_inf = (str3ncmp_case_insensitive(src, "inf") | (src[3] ^ '"')) == 0;
+  if(is_short_inf) return true;
+
+  return (str8ncmp_case_insensitive(src, "infinity") | (src[8] ^ '"')) == 0;
+}
+
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src, size_t len) {
+  if (len > 8) { return is_valid_inf_atom(src); }
+  if (len == 8 && str8ncmp_case_insensitive(src, "infinity") == 0) {
+    return true;
+  }
+  if (len > 3) {
+    return (str3ncmp_case_insensitive(src, "inf")
+          | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+  }
+  if (len == 3) { return str3ncmp_case_insensitive(src, "inf") == 0; }
+  return false;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
 } // namespace atomparsing
 } // unnamed namespace
 } // namespace lasx
@@ -31255,7 +36474,7 @@ inline simdjson_warn_unused error_code dom_parser_implementation::set_capacity(s
 }

 inline simdjson_warn_unused error_code dom_parser_implementation::set_max_depth(size_t max_depth) noexcept {
-  if(max_depth == 0 || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
+  if(max_depth == 0 || max_depth > SIMDJSON_MAX_DEPTH || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
   // Stage 2 stacks
   open_containers.reset(new (std::nothrow) open_container[max_depth]);
   is_array.reset(new (std::nothrow) bool[max_depth]);
@@ -31434,6 +36653,7 @@ protected:
 /* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_NUMBERPARSING_H */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/base.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/jsoncharutils.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/atomparsing.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/internal/numberparsing_tables.h" */
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */

@@ -31473,6 +36693,18 @@ simdjson_inline double to_double(uint64_t mantissa, uint64_t real_exponent, bool
     return d;
 }

+// Convert a mantissa, an exponent and a sign bit into an ieee32 float (binary32).
+// The real_exponent needs to be in [0, 254] (technically real_exponent = 255 would be acceptable).
+// The mantissa should be in [0,1<<24). The bit at index (1U << 23) will be zeroed.
+simdjson_inline float to_float(uint32_t mantissa, uint32_t real_exponent, bool negative) {
+    float f;
+    mantissa &= ~(uint32_t(1) << 23);
+    mantissa |= real_exponent << 23;
+    mantissa |= ((static_cast<uint32_t>(negative)) << 31);
+    std::memcpy(&f, &mantissa, sizeof(f));
+    return f;
+}
+
 // Attempts to compute i * 10^(power) exactly; and if "negative" is
 // true, negate the result.
 // This function will only work in some cases, when it does not work, success is
@@ -31729,6 +36961,219 @@ simdjson_inline bool compute_float_64(int64_t power, uint64_t i, bool negative,
   return true;
 }

+// Attempts to compute i * 10^(power) as a binary32 (float) value; and if
+// "negative" is true, negate the result.
+//
+// This is the same Eisel-Lemire algorithm as compute_float_64, retargeted at
+// binary32. It is adapted from fast_float
+// (https://github.com/fastfloat/fast_float), where the routine is written once
+// and instantiated for each binary format. We only need 24 bits of mantissa
+// instead of 53, so we ask the truncated product for 24 + 2 = 26 accurate bits
+// (plus one bit that may be lost to the "upperbit" shift), and we shift the
+// product right by (upperbit + 64 - 23 - 3) instead of (upperbit + 64 - 52 - 3).
+//
+// The power-of-five table is shared with the binary64 code: the useful range
+// for binary32, [smallest_power_binary32, largest_power_binary32], is a strict
+// subset of [smallest_power, largest_power].
+//
+// The function returns false when the result would be infinite: simdjson
+// refuses to parse infinite values, so the caller reports an error. Unlike
+// compute_float_64, a false return never means "try harder": the accuracy of
+// the two-limb product is guaranteed by Noble Mushtak and Daniel Lemire,
+// "Fast Number Parsing Without Fallback" (https://arxiv.org/abs/2212.06644).
+// The caller must still fall back to a slow path when the 64-bit mantissa i
+// was truncated (more than 19 significant digits).
+//
+// We assume that power is in the [smallest_power, largest_power] interval: the
+// caller is responsible for this check.
+simdjson_inline bool compute_float_32(int64_t power, uint64_t i, bool negative, float &d) {
+  // Powers of ten that are exactly representable as binary32 values: 10^k is
+  // exact as long as k <= 10 (5^10 = 9765625 < 2**24 while 5^11 > 2**24).
+  static constexpr float power_of_ten_float[] = {
+      1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+      1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+  // The range of powers of ten that a non-zero, finite binary32 value can be
+  // built from. Anything smaller rounds to zero, anything larger is infinite.
+  // These are the smallest_power_of_ten()/largest_power_of_ten() constants that
+  // fast_float uses for binary32.
+  constexpr int smallest_power_binary32 = -65;
+  constexpr int largest_power_binary32 = 38;
+
+  // We start with the fast path described in
+  // Clinger WD. How to read floating point numbers accurately.
+  // ACM SIGPLAN Notices. 1990
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+  // We cannot be certain that x/y is rounded to nearest.
+  if (0 <= power && power <= 10 && i <= 16777215)
+#else
+  if (-10 <= power && power <= 10 && i <= 16777215)
+#endif
+  {
+    // Convert the integer into a float. This is lossless since
+    // 0 <= i <= 2^24 - 1.
+    d = float(i);
+    // Both d and the power of ten are exactly representable as binary32
+    // values, so the product (or quotient) is correctly rounded.
+    if (power < 0) {
+      d = d / power_of_ten_float[-power];
+    } else {
+      d = d * power_of_ten_float[power];
+    }
+    if (negative) {
+      d = -d;
+    }
+    return true;
+  }
+
+  // The fast path has failed, so we fall back on the Eisel-Lemire algorithm.
+  // It needs i > 0 (so that the leading bit of i can be normalized), so we
+  // handle i == 0 separately. We also handle the powers of ten that are so
+  // small (or so large) that the answer is zero (or infinite) whatever the
+  // mantissa is.
+  if (i == 0 || power < smallest_power_binary32) {
+    d = negative ? -0.0f : 0.0f;
+    return true;
+  }
+  if (power > largest_power_binary32) {
+    // We have, for sure, an infinite value.
+    return false;
+  }
+
+  // The exponent is 128 + 63 + power + floor(log(5**power)/log(2)).
+  // The 128 comes from the ieee32 standard (the minimal exponent is -127).
+  // The 63 comes from the fact that we use a 64-bit word.
+  // See compute_float_64 for a discussion of the magical 152170 + 65536.
+  int64_t exponent = (((152170 + 65536) * power) >> 16) + 128 + 63;
+
+  // We want the most significant bit of i to be 1. Shift if needed.
+  int lz = leading_zeroes(i);
+  i <<= lz;
+
+  // We want the most significant 64 bits of the product i * 5**power. It is
+  // safe to index the table because
+  // smallest_power <= smallest_power_binary32 <= power
+  //               <= largest_power_binary32 <= largest_power.
+  const uint32_t index = 2 * uint32_t(power - simdjson::internal::smallest_power);
+#if SIMDJSON_STATIC_REFLECTION
+  simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index]);
+#else
+  simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index]);
+#endif
+
+  // Unless the least significant 38 bits of the high (64-bit) part of the full
+  // product are all 1s, then we know that the most significant 26 bits are
+  // exact and no further work is needed. Having 26 bits is necessary because
+  // we need 24 bits for the mantissa but we have to have one rounding bit and
+  // we can waste a bit if the most significant bit of the product is zero.
+  // (For binary64, the same reasoning gives 55 bits and a 9-bit mask.)
+  if((firstproduct.high & 0x3FFFFFFFFF) == 0x3FFFFFFFFF) {
+    // The truncated multiplication was not accurate enough; use the next 64
+    // bits of the power of five to refine it. See compute_float_64 for a
+    // detailed discussion.
+#if SIMDJSON_STATIC_REFLECTION
+    simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index + 1]);
+#else
+    simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index + 1]);
+#endif
+    firstproduct.low += secondproduct.high;
+    if(secondproduct.high > firstproduct.low) { firstproduct.high++; }
+  }
+  uint64_t lower = firstproduct.low;
+  uint64_t upper = firstproduct.high;
+  // The final mantissa should be 24 bits with a leading 1.
+  // We shift it so that it occupies 25 bits with a leading 1.
+  ///////
+  uint64_t upperbit = upper >> 63;
+  uint64_t mantissa = upper >> (upperbit + 38); // 38 == 64 - 23 - 3
+  lz += int(1 ^ upperbit);
+
+  // Here we have mantissa < (1<<25).
+  int64_t real_exponent = exponent - lz;
+  if (simdjson_unlikely(real_exponent <= 0)) { // we have a subnormal?
+    // Here we have that real_exponent <= 0 so -real_exponent >= 0
+    if(-real_exponent + 1 >= 64) { // if we have more than 64 bits below the minimum exponent, you have a zero for sure.
+      d = negative ? -0.0f : 0.0f;
+      return true;
+    }
+    // next line is safe because -real_exponent + 1 < 64
+    mantissa >>= -real_exponent + 1;
+    // Thankfully, we can't have both "round-to-even" and subnormals because
+    // "round-to-even" only occurs for powers close to 0.
+    mantissa += (mantissa & 1); // round up
+    mantissa >>= 1;
+    // As in compute_float_64, rounding up may take us out of the subnormal
+    // range, so we can only decide after rounding.
+    real_exponent = (mantissa < (uint64_t(1) << 23)) ? 0 : 1;
+    d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+    return true;
+  }
+  // We have to round to even. The "to even" part is only a problem when we are
+  // right in between two floats, which we guard against. The bounds on the
+  // power of ten are those of fast_float's
+  // min_exponent_round_to_even()/max_exponent_round_to_even() for binary32:
+  // when q >= 0, (2m+1) must be divisible by 5^q with 5^q <= 2^25, so q <= 10;
+  // when q < 0, we need 2^24 x 5^{-q} < 2^{64}, so q >= -17.
+  if (simdjson_unlikely((lower <= 1) && (power >= -17) && (power <= 10) && ((mantissa & 3) == 1))) {
+    if((mantissa << (upperbit + 38)) == upper) {
+      mantissa &= ~uint64_t(1);   // flip it so that we do not round up
+    }
+  }
+
+  mantissa += mantissa & 1;
+  mantissa >>= 1;
+
+  // Here we have mantissa < (1<<24), unless there was an overflow
+  if (mantissa >= (uint64_t(1) << 24)) {
+    mantissa = (uint64_t(1) << 23);
+    real_exponent++;
+  }
+  mantissa &= ~(uint64_t(1) << 23);
+  // we have to check that real_exponent is in range, otherwise we bail out
+  if (simdjson_unlikely(real_exponent > 254)) {
+    // We have an infinite value!!! We could actually throw an error here if we could.
+    return false;
+  }
+  d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+  return true;
+}
+
+#if SIMDJSON_ENABLE_NAN_INF
+// Parses a nan or infinity. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, double& d) noexcept {
+  if (atomparsing::is_valid_inf_atom(src)) {
+    double inf = std::numeric_limits<double>::infinity();
+    d = negative ? -inf : inf;
+    return true;
+  }
+
+  if (atomparsing::is_valid_nan_atom(src)) {
+    d = std::numeric_limits<double>::quiet_NaN();
+    return true;
+  }
+
+  return false;
+}
+
+// Parses a nan or infinity as a binary32 value. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, float& d) noexcept {
+  if (atomparsing::is_valid_inf_atom(src)) {
+    float inf = std::numeric_limits<float>::infinity();
+    d = negative ? -inf : inf;
+    return true;
+  }
+
+  if (atomparsing::is_valid_nan_atom(src)) {
+    d = std::numeric_limits<float>::quiet_NaN();
+    return true;
+  }
+
+  return false;
+}
+#endif
+
 // We call a fallback floating-point parser that might be slow. Note
 // it will accept JSON numbers, but the JSON spec. is more restrictive so
 // before you call parse_float_fallback, you need to have validated the input
@@ -31766,6 +37211,16 @@ static bool parse_float_fallback(const uint8_t *ptr, const uint8_t *end_ptr, dou
   return !(*outDouble > (std::numeric_limits<double>::max)() || *outDouble < std::numeric_limits<double>::lowest());
 }

+// Same as parse_float_fallback, but for binary32 (float) values. Going through
+// the binary64 fallback and then rounding to binary32 would be subject to
+// double rounding, so we run the fallback algorithm directly on binary32.
+static bool parse_float_fallback(const uint8_t *ptr, float *outFloat) {
+  *outFloat = simdjson::internal::from_chars_float(reinterpret_cast<const char *>(ptr));
+  // We do not accept infinite values. See the binary64 version above for why we
+  // do not use std::isfinite.
+  return !(*outFloat > (std::numeric_limits<float>::max)() || *outFloat < std::numeric_limits<float>::lowest());
+}
+
 // check quickly whether the next 8 chars are made of digits
 // at a glance, it looks better than Mula's
 // http://0x80.pl/articles/swar-digits-validate.html
@@ -31784,6 +37239,28 @@ simdjson_inline bool is_made_of_eight_digits_fast(const uint8_t *chars) {
           0x3333333333333333);
 }

+// The same idea for four characters. A block of eight only bites when eight
+// digits are left, so a run of 4 to 7 digits used to fall back to one digit at a
+// time; taking four of them at once is what makes the tail of a long fraction
+// cheap. Adapted from fast_float, credit @aqrit.
+simdjson_inline bool is_made_of_four_digits_fast(const uint8_t *chars) {
+  uint32_t val;
+  // Reads up to 3 bytes beyond the digits, which SIMDJSON_PADDING covers.
+  static_assert(3 <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be bigger than 3");
+  std::memcpy(&val, chars, 4);
+  return !(((val + 0x46464646) | (val - 0x30303030)) & 0x80808080);
+}
+
+// Only call this when is_made_of_four_digits_fast() says the four characters are
+// digits. Little-endian: guarded by SIMDJSON_SWAR_NUMBER_PARSING.
+simdjson_inline uint32_t parse_four_digits_unrolled(const uint8_t *chars) {
+  uint32_t val;
+  std::memcpy(&val, chars, 4);
+  val -= 0x30303030;
+  val = (val * 10) + (val >> 8);
+  return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
 template<typename I>
 SIMDJSON_NO_SANITIZE_UNDEFINED // We deliberately allow overflow here and check later
 simdjson_inline bool parse_digit(const uint8_t c, I &i) {
@@ -31800,26 +37277,73 @@ simdjson_inline bool is_digit(const uint8_t c) {
   return static_cast<uint8_t>(c - '0') <= 9;
 }

-simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
-  // we continue with the fiction that we have an integer. If the
-  // floating point number is representable as x * 10^z for some integer
-  // z that fits in 53 bits, then we will be able to convert back the
-  // the integer into a float in a lossless manner.
-  const uint8_t *const first_after_period = p;
+// Consumes a run of digits into i, eight at a time while eight are available,
+// then four, then one at a time. Overflow is deliberate: the caller counts the
+// digits and falls back when there are too many for a 64-bit mantissa.
+//
+// Long runs of digits are what a float-heavy document is made of: the fraction
+// of a binary64 printed to full precision is 15 to 17 digits, and reading those
+// one at a time is the single largest cost in parse_double(). Integer parts are
+// left alone: they are usually a handful of digits, and there the check for a
+// block of eight is wasted work.
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_fraction_digits(const uint8_t *&p, uint64_t &i) {
+#ifdef SIMDJSON_SWAR_NUMBER_PARSING
+#if SIMDJSON_SWAR_NUMBER_PARSING
+  while (is_made_of_eight_digits_fast(p)) {
+    i = i * 100000000 + parse_eight_digits_unrolled(p);
+    p += 8;
+  }
+  // A 4 to 7 digit remainder is the common case once the blocks of eight are
+  // gone: 15 fraction digits leave 7, and 17 leave 1 after two blocks.
+  if (is_made_of_four_digits_fast(p)) {
+    i = i * 10000 + parse_four_digits_unrolled(p);
+    p += 4;
+  }
+#endif // SIMDJSON_SWAR_NUMBER_PARSING
+#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
+  while (parse_digit(*p, i)) { p++; }
+}
+
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_float_integer_digits(const uint8_t *&p, uint64_t &i,
+                                                bool &leading_zero) {
+  if (!parse_digit(*p, i)) { leading_zero = true; return; }
+  p++;
+  leading_zero = (i == 0);
+  while (parse_digit(*p, i)) { p++; }
+}

+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_integer_digits(const uint8_t *&p, uint64_t &i) {
 #ifdef SIMDJSON_SWAR_NUMBER_PARSING
 #if SIMDJSON_SWAR_NUMBER_PARSING
-  // this helps if we have lots of decimals!
-  // this turns out to be frequent enough.
+  // Identifiers, timestamps and counters often have eight digits or more.
+  // Prior related work: jsonifier parses integers as eight-digit SWAR words
+  // (str_to_i.hpp, https://github.com/nihilai-collective/Jsonifier). This
+  // takes one such word with parse_eight_digits_unrolled, the routine
+  // simdjson uses for long fractions.
   if (is_made_of_eight_digits_fast(p)) {
     i = i * 100000000 + parse_eight_digits_unrolled(p);
     p += 8;
   }
+  const uint8_t *const swar_end = p + 8;
+  while (p < swar_end && is_made_of_four_digits_fast(p)) {
+    i = i * 10000 + parse_four_digits_unrolled(p);
+    p += 4;
+  }
 #endif // SIMDJSON_SWAR_NUMBER_PARSING
 #endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
-  // Unrolling the first digit makes a small difference on some implementations (e.g. westmere)
-  if (parse_digit(*p, i)) { ++p; }
   while (parse_digit(*p, i)) { p++; }
+}
+
+simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
+  // we continue with the fiction that we have an integer. If the
+  // floating point number is representable as x * 10^z for some integer
+  // z that fits in 53 bits, then we will be able to convert back the
+  // the integer into a float in a lossless manner.
+  const uint8_t *const first_after_period = p;
+  parse_fraction_digits(p, i);
   exponent = first_after_period - p;
   // Decimal without digits (123.) is illegal
   if (exponent == 0) {
@@ -31908,7 +37432,7 @@ simdjson_inline size_t significant_digits(const uint8_t * start_digits, size_t d
 } // unnamed namespace

 /** @private */
-static error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
+inline error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
   if (parse_float_fallback(src, answer)) {
     return SUCCESS;
   }
@@ -31991,15 +37515,17 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
   return SUCCESS;              // always succeeds
 }

-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept  { return false; }
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept  { return false; }
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept { return number_type::signed_integer; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline bool is_negative(const uint8_t *) noexcept  { return false; }
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t *) noexcept  { return false; }
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t *) noexcept { return number_type::signed_integer; }
 #else

 // parse the number at src
@@ -32030,7 +37556,21 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
   size_t digit_count = size_t(p - start_digits);
-  if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) { return INVALID_NUMBER(src); }
+  if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // By this point, we know that our input does not begin with a digit. We will attempt
+    // to handle NaN/Infinity.
+
+    double d;
+    if (compute_nan_inf(p, negative, d)) {
+      writer.append_double(d);
+      return SUCCESS;
+    }
+#endif
+
+    return INVALID_NUMBER(src);
+  }

   //
   // Handle floats if there is a . or e (or both)
@@ -32079,7 +37619,7 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
     // - Therefore, if the number is positive and lower than that, it's overflow.
     // - The value we are looking at is less than or equal to INT64_MAX.
     //
-    }  else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return INVALID_NUMBER(src); }
+    }  else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return BIGINT_NUMBER(src); }
   }

   // Write unsigned if it does not fit in a signed integer.
@@ -32178,7 +37718,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const u
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -32276,7 +37816,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_stri
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -32331,7 +37871,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uin
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -32417,7 +37957,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = src;
   uint64_t i = 0;
-  while (parse_digit(*src, i)) { src++; }
+  parse_integer_digits(src, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -32457,11 +37997,20 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   //
   uint64_t i = 0;
   const uint8_t *p = src;
-  p += parse_digit(*p, i);
-  bool leading_zero = (i == 0);
-  while (parse_digit(*p, i)) { p++; }
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
   // no integer digits, or 0123 (zero must be solo)
-  if ( p == src ) { return INCORRECT_TYPE; }
+  if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no loading digits, the number may be nan or infinity.
+    // Attempt to compute those, and return on success.
+    double d;
+    if (compute_nan_inf(p, negative, d)) { return d; }
+#endif
+
+    return INCORRECT_TYPE;
+  }
   if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }

   //
@@ -32472,9 +38021,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   if (simdjson_likely(*p == '.')) {
     p++;
     const uint8_t *start_decimal_digits = p;
-    if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
-    p++;
-    while (parse_digit(*p, i)) { p++; }
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
     exponent = -(p - start_decimal_digits);

     // Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -32523,6 +38071,97 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   return d;
 }

+// Parse a JSON number into a binary32 (float) value.
+//
+// This mirrors parse_double, but it rounds to binary32 directly instead of
+// rounding to binary64 and then to binary32: the latter is subject to double
+// rounding and would not always produce the nearest float.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * src) noexcept {
+  //
+  // Check for minus sign
+  //
+  bool negative = (*src == '-');
+  src += uint8_t(negative);
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, the number may be nan or infinity.
+    // Attempt to compute those, and return on success.
+    float f;
+    if (compute_nan_inf(p, negative, f)) { return f; }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = p-start_digits > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if (*p == 'e' || *p == 'E') {
+    p++;
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while (parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  float d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_32(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
 simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
   return (*src == '-');
 }
@@ -32675,11 +38314,25 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   //
   uint64_t i = 0;
   const uint8_t *p = src;
-  p += parse_digit(*p, i);
-  bool leading_zero = (i == 0);
-  while (parse_digit(*p, i)) { p++; }
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
   // no integer digits, or 0123 (zero must be solo)
-  if ( p == src ) { return INCORRECT_TYPE; }
+  if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, attempt to parse numbers that are either
+    // NaN or Infinity
+    if (atomparsing::is_valid_inf_in_string(src)) {
+      double inf = std::numeric_limits<double>::infinity();
+      return negative ? -inf : inf;
+    }
+
+    if (atomparsing::is_valid_nan_in_string(src)) {
+      return std::numeric_limits<double>::quiet_NaN();
+    }
+#endif
+
+    return INCORRECT_TYPE;
+  }
   if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }

   //
@@ -32690,9 +38343,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   if (simdjson_likely(*p == '.')) {
     p++;
     const uint8_t *start_decimal_digits = p;
-    if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
-    p++;
-    while (parse_digit(*p, i)) { p++; }
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
     exponent = -(p - start_decimal_digits);

     // Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -32741,6 +38393,99 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   return d;
 }

+// Parse a JSON number held inside a JSON string into a binary32 (float) value.
+// See parse_float for why we do not simply round parse_double_in_string.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * src) noexcept {
+  //
+  // Check for minus sign
+  //
+  bool negative = (*(src + 1) == '-');
+  src += uint8_t(negative) + 1;
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, attempt to parse numbers that are either
+    // NaN or Infinity
+    if (atomparsing::is_valid_inf_in_string(src)) {
+      float inf = std::numeric_limits<float>::infinity();
+      return negative ? -inf : inf;
+    }
+
+    if (atomparsing::is_valid_nan_in_string(src)) {
+      return std::numeric_limits<float>::quiet_NaN();
+    }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = p-start_digits > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if (*p == 'e' || *p == 'E') {
+    p++;
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while (parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if (*p != '"') { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  float d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_32(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
 } // unnamed namespace
 #endif // SIMDJSON_SKIPNUMBERPARSING

@@ -33002,10 +38747,6 @@ simdjson_inline int count_ones(uint64_t input_num) {
   return __lsx_vpickve2gr_w(__lsx_vpcnt_d(__m128i(v2u64{input_num, 0})), 0);
 }

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2, uint64_t *result) {
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-}

 } // unnamed namespace
 } // namespace lsx
@@ -33734,6 +39475,9 @@ namespace atomparsing {
 // to the compile-time constant 1936482662.
 simdjson_inline uint32_t string_to_uint32(const char* str) { uint32_t val; std::memcpy(&val, str, sizeof(uint32_t)); return val; }

+// Acts on the same principle as string_to_uint32, but on an 8-byte block of memory
+simdjson_inline uint64_t string_to_uint64(const char* str) { uint64_t val; std::memcpy(&val, str, sizeof(uint64_t)); return val; }
+

 // Again in str4ncmp we use a memcpy to avoid undefined behavior. The memcpy may appear expensive.
 // Yet all decent optimizing compilers will compile memcpy to a single instruction, just about.
@@ -33745,6 +39489,28 @@ simdjson_inline uint32_t str4ncmp(const uint8_t *src, const char* atom) {
   return srcval ^ string_to_uint32(atom);
 }

+// Checks that the first 8 characters of the input string match the given atom in a case-insensitive manner.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint64_t str8ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+  uint64_t srcval; // we want to avoid unaligned 32-bit loads (undefined in C/C++)
+  static_assert(sizeof(uint64_t) <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be larger than 8 bytes");
+  std::memcpy(&srcval, src, sizeof(uint64_t));
+
+  return (srcval | 0x2020202020202020ull) ^ string_to_uint64(atom);
+}
+
+// Checks that the first 3 characters of 'src' match 'atom' in a case-insensitive way.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint32_t str3ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+  return ((src[0] | 0x20) ^ atom[0]) //
+       | ((src[1] | 0x20) ^ atom[1]) //
+       | ((src[2] | 0x20) ^ atom[2]);
+}
+
 simdjson_warn_unused
 simdjson_inline bool is_valid_true_atom(const uint8_t *src) {
   return (str4ncmp(src, "true") | jsoncharutils::is_not_structural_or_whitespace(src[4])) == 0;
@@ -33781,6 +39547,71 @@ simdjson_inline bool is_valid_null_atom(const uint8_t *src, size_t len) {
   else { return false; }
 }

+#if SIMDJSON_ENABLE_NAN_INF
+// "nan" is 3 bytes; we check characters and then verify the next
+// character is structural or whitespace. We accept both "nan" and "NaN".
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src) {
+  return (str3ncmp_case_insensitive(src, "nan")
+        | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+}
+
+// checks that the next four characters of a string are 'nan"', where the 'nan'
+// is checked in a case-insensitive way.
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_in_string(const uint8_t *src) {
+  return (str3ncmp_case_insensitive(src, "nan") | (src[3] ^ '"')) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src, size_t len) {
+  if (len > 3) { return is_valid_nan_atom(src); }
+  if (len == 3) { return str3ncmp_case_insensitive(src, "nan") == 0; }
+  return false;
+}
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src) {
+  bool is_short_inf = (str3ncmp_case_insensitive(src, "inf")
+                     | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+  if(is_short_inf) return true;
+
+  // Check for 'infinity' (any capitalization)
+  return (str8ncmp_case_insensitive(src, "infinity") | jsoncharutils::is_not_structural_or_whitespace(src[8])) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_in_string(const uint8_t *src) {
+  bool is_short_inf = (str3ncmp_case_insensitive(src, "inf") | (src[3] ^ '"')) == 0;
+  if(is_short_inf) return true;
+
+  return (str8ncmp_case_insensitive(src, "infinity") | (src[8] ^ '"')) == 0;
+}
+
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src, size_t len) {
+  if (len > 8) { return is_valid_inf_atom(src); }
+  if (len == 8 && str8ncmp_case_insensitive(src, "infinity") == 0) {
+    return true;
+  }
+  if (len > 3) {
+    return (str3ncmp_case_insensitive(src, "inf")
+          | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+  }
+  if (len == 3) { return str3ncmp_case_insensitive(src, "inf") == 0; }
+  return false;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
 } // namespace atomparsing
 } // unnamed namespace
 } // namespace lsx
@@ -33871,7 +39702,7 @@ inline simdjson_warn_unused error_code dom_parser_implementation::set_capacity(s
 }

 inline simdjson_warn_unused error_code dom_parser_implementation::set_max_depth(size_t max_depth) noexcept {
-  if(max_depth == 0 || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
+  if(max_depth == 0 || max_depth > SIMDJSON_MAX_DEPTH || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
   // Stage 2 stacks
   open_containers.reset(new (std::nothrow) open_container[max_depth]);
   is_array.reset(new (std::nothrow) bool[max_depth]);
@@ -34050,6 +39881,7 @@ protected:
 /* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_NUMBERPARSING_H */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/base.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/jsoncharutils.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/atomparsing.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/internal/numberparsing_tables.h" */
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */

@@ -34089,6 +39921,18 @@ simdjson_inline double to_double(uint64_t mantissa, uint64_t real_exponent, bool
     return d;
 }

+// Convert a mantissa, an exponent and a sign bit into an ieee32 float (binary32).
+// The real_exponent needs to be in [0, 254] (technically real_exponent = 255 would be acceptable).
+// The mantissa should be in [0,1<<24). The bit at index (1U << 23) will be zeroed.
+simdjson_inline float to_float(uint32_t mantissa, uint32_t real_exponent, bool negative) {
+    float f;
+    mantissa &= ~(uint32_t(1) << 23);
+    mantissa |= real_exponent << 23;
+    mantissa |= ((static_cast<uint32_t>(negative)) << 31);
+    std::memcpy(&f, &mantissa, sizeof(f));
+    return f;
+}
+
 // Attempts to compute i * 10^(power) exactly; and if "negative" is
 // true, negate the result.
 // This function will only work in some cases, when it does not work, success is
@@ -34345,6 +40189,219 @@ simdjson_inline bool compute_float_64(int64_t power, uint64_t i, bool negative,
   return true;
 }

+// Attempts to compute i * 10^(power) as a binary32 (float) value; and if
+// "negative" is true, negate the result.
+//
+// This is the same Eisel-Lemire algorithm as compute_float_64, retargeted at
+// binary32. It is adapted from fast_float
+// (https://github.com/fastfloat/fast_float), where the routine is written once
+// and instantiated for each binary format. We only need 24 bits of mantissa
+// instead of 53, so we ask the truncated product for 24 + 2 = 26 accurate bits
+// (plus one bit that may be lost to the "upperbit" shift), and we shift the
+// product right by (upperbit + 64 - 23 - 3) instead of (upperbit + 64 - 52 - 3).
+//
+// The power-of-five table is shared with the binary64 code: the useful range
+// for binary32, [smallest_power_binary32, largest_power_binary32], is a strict
+// subset of [smallest_power, largest_power].
+//
+// The function returns false when the result would be infinite: simdjson
+// refuses to parse infinite values, so the caller reports an error. Unlike
+// compute_float_64, a false return never means "try harder": the accuracy of
+// the two-limb product is guaranteed by Noble Mushtak and Daniel Lemire,
+// "Fast Number Parsing Without Fallback" (https://arxiv.org/abs/2212.06644).
+// The caller must still fall back to a slow path when the 64-bit mantissa i
+// was truncated (more than 19 significant digits).
+//
+// We assume that power is in the [smallest_power, largest_power] interval: the
+// caller is responsible for this check.
+simdjson_inline bool compute_float_32(int64_t power, uint64_t i, bool negative, float &d) {
+  // Powers of ten that are exactly representable as binary32 values: 10^k is
+  // exact as long as k <= 10 (5^10 = 9765625 < 2**24 while 5^11 > 2**24).
+  static constexpr float power_of_ten_float[] = {
+      1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+      1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+  // The range of powers of ten that a non-zero, finite binary32 value can be
+  // built from. Anything smaller rounds to zero, anything larger is infinite.
+  // These are the smallest_power_of_ten()/largest_power_of_ten() constants that
+  // fast_float uses for binary32.
+  constexpr int smallest_power_binary32 = -65;
+  constexpr int largest_power_binary32 = 38;
+
+  // We start with the fast path described in
+  // Clinger WD. How to read floating point numbers accurately.
+  // ACM SIGPLAN Notices. 1990
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+  // We cannot be certain that x/y is rounded to nearest.
+  if (0 <= power && power <= 10 && i <= 16777215)
+#else
+  if (-10 <= power && power <= 10 && i <= 16777215)
+#endif
+  {
+    // Convert the integer into a float. This is lossless since
+    // 0 <= i <= 2^24 - 1.
+    d = float(i);
+    // Both d and the power of ten are exactly representable as binary32
+    // values, so the product (or quotient) is correctly rounded.
+    if (power < 0) {
+      d = d / power_of_ten_float[-power];
+    } else {
+      d = d * power_of_ten_float[power];
+    }
+    if (negative) {
+      d = -d;
+    }
+    return true;
+  }
+
+  // The fast path has failed, so we fall back on the Eisel-Lemire algorithm.
+  // It needs i > 0 (so that the leading bit of i can be normalized), so we
+  // handle i == 0 separately. We also handle the powers of ten that are so
+  // small (or so large) that the answer is zero (or infinite) whatever the
+  // mantissa is.
+  if (i == 0 || power < smallest_power_binary32) {
+    d = negative ? -0.0f : 0.0f;
+    return true;
+  }
+  if (power > largest_power_binary32) {
+    // We have, for sure, an infinite value.
+    return false;
+  }
+
+  // The exponent is 128 + 63 + power + floor(log(5**power)/log(2)).
+  // The 128 comes from the ieee32 standard (the minimal exponent is -127).
+  // The 63 comes from the fact that we use a 64-bit word.
+  // See compute_float_64 for a discussion of the magical 152170 + 65536.
+  int64_t exponent = (((152170 + 65536) * power) >> 16) + 128 + 63;
+
+  // We want the most significant bit of i to be 1. Shift if needed.
+  int lz = leading_zeroes(i);
+  i <<= lz;
+
+  // We want the most significant 64 bits of the product i * 5**power. It is
+  // safe to index the table because
+  // smallest_power <= smallest_power_binary32 <= power
+  //               <= largest_power_binary32 <= largest_power.
+  const uint32_t index = 2 * uint32_t(power - simdjson::internal::smallest_power);
+#if SIMDJSON_STATIC_REFLECTION
+  simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index]);
+#else
+  simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index]);
+#endif
+
+  // Unless the least significant 38 bits of the high (64-bit) part of the full
+  // product are all 1s, then we know that the most significant 26 bits are
+  // exact and no further work is needed. Having 26 bits is necessary because
+  // we need 24 bits for the mantissa but we have to have one rounding bit and
+  // we can waste a bit if the most significant bit of the product is zero.
+  // (For binary64, the same reasoning gives 55 bits and a 9-bit mask.)
+  if((firstproduct.high & 0x3FFFFFFFFF) == 0x3FFFFFFFFF) {
+    // The truncated multiplication was not accurate enough; use the next 64
+    // bits of the power of five to refine it. See compute_float_64 for a
+    // detailed discussion.
+#if SIMDJSON_STATIC_REFLECTION
+    simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index + 1]);
+#else
+    simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index + 1]);
+#endif
+    firstproduct.low += secondproduct.high;
+    if(secondproduct.high > firstproduct.low) { firstproduct.high++; }
+  }
+  uint64_t lower = firstproduct.low;
+  uint64_t upper = firstproduct.high;
+  // The final mantissa should be 24 bits with a leading 1.
+  // We shift it so that it occupies 25 bits with a leading 1.
+  ///////
+  uint64_t upperbit = upper >> 63;
+  uint64_t mantissa = upper >> (upperbit + 38); // 38 == 64 - 23 - 3
+  lz += int(1 ^ upperbit);
+
+  // Here we have mantissa < (1<<25).
+  int64_t real_exponent = exponent - lz;
+  if (simdjson_unlikely(real_exponent <= 0)) { // we have a subnormal?
+    // Here we have that real_exponent <= 0 so -real_exponent >= 0
+    if(-real_exponent + 1 >= 64) { // if we have more than 64 bits below the minimum exponent, you have a zero for sure.
+      d = negative ? -0.0f : 0.0f;
+      return true;
+    }
+    // next line is safe because -real_exponent + 1 < 64
+    mantissa >>= -real_exponent + 1;
+    // Thankfully, we can't have both "round-to-even" and subnormals because
+    // "round-to-even" only occurs for powers close to 0.
+    mantissa += (mantissa & 1); // round up
+    mantissa >>= 1;
+    // As in compute_float_64, rounding up may take us out of the subnormal
+    // range, so we can only decide after rounding.
+    real_exponent = (mantissa < (uint64_t(1) << 23)) ? 0 : 1;
+    d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+    return true;
+  }
+  // We have to round to even. The "to even" part is only a problem when we are
+  // right in between two floats, which we guard against. The bounds on the
+  // power of ten are those of fast_float's
+  // min_exponent_round_to_even()/max_exponent_round_to_even() for binary32:
+  // when q >= 0, (2m+1) must be divisible by 5^q with 5^q <= 2^25, so q <= 10;
+  // when q < 0, we need 2^24 x 5^{-q} < 2^{64}, so q >= -17.
+  if (simdjson_unlikely((lower <= 1) && (power >= -17) && (power <= 10) && ((mantissa & 3) == 1))) {
+    if((mantissa << (upperbit + 38)) == upper) {
+      mantissa &= ~uint64_t(1);   // flip it so that we do not round up
+    }
+  }
+
+  mantissa += mantissa & 1;
+  mantissa >>= 1;
+
+  // Here we have mantissa < (1<<24), unless there was an overflow
+  if (mantissa >= (uint64_t(1) << 24)) {
+    mantissa = (uint64_t(1) << 23);
+    real_exponent++;
+  }
+  mantissa &= ~(uint64_t(1) << 23);
+  // we have to check that real_exponent is in range, otherwise we bail out
+  if (simdjson_unlikely(real_exponent > 254)) {
+    // We have an infinite value!!! We could actually throw an error here if we could.
+    return false;
+  }
+  d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+  return true;
+}
+
+#if SIMDJSON_ENABLE_NAN_INF
+// Parses a nan or infinity. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, double& d) noexcept {
+  if (atomparsing::is_valid_inf_atom(src)) {
+    double inf = std::numeric_limits<double>::infinity();
+    d = negative ? -inf : inf;
+    return true;
+  }
+
+  if (atomparsing::is_valid_nan_atom(src)) {
+    d = std::numeric_limits<double>::quiet_NaN();
+    return true;
+  }
+
+  return false;
+}
+
+// Parses a nan or infinity as a binary32 value. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, float& d) noexcept {
+  if (atomparsing::is_valid_inf_atom(src)) {
+    float inf = std::numeric_limits<float>::infinity();
+    d = negative ? -inf : inf;
+    return true;
+  }
+
+  if (atomparsing::is_valid_nan_atom(src)) {
+    d = std::numeric_limits<float>::quiet_NaN();
+    return true;
+  }
+
+  return false;
+}
+#endif
+
 // We call a fallback floating-point parser that might be slow. Note
 // it will accept JSON numbers, but the JSON spec. is more restrictive so
 // before you call parse_float_fallback, you need to have validated the input
@@ -34382,6 +40439,16 @@ static bool parse_float_fallback(const uint8_t *ptr, const uint8_t *end_ptr, dou
   return !(*outDouble > (std::numeric_limits<double>::max)() || *outDouble < std::numeric_limits<double>::lowest());
 }

+// Same as parse_float_fallback, but for binary32 (float) values. Going through
+// the binary64 fallback and then rounding to binary32 would be subject to
+// double rounding, so we run the fallback algorithm directly on binary32.
+static bool parse_float_fallback(const uint8_t *ptr, float *outFloat) {
+  *outFloat = simdjson::internal::from_chars_float(reinterpret_cast<const char *>(ptr));
+  // We do not accept infinite values. See the binary64 version above for why we
+  // do not use std::isfinite.
+  return !(*outFloat > (std::numeric_limits<float>::max)() || *outFloat < std::numeric_limits<float>::lowest());
+}
+
 // check quickly whether the next 8 chars are made of digits
 // at a glance, it looks better than Mula's
 // http://0x80.pl/articles/swar-digits-validate.html
@@ -34400,6 +40467,28 @@ simdjson_inline bool is_made_of_eight_digits_fast(const uint8_t *chars) {
           0x3333333333333333);
 }

+// The same idea for four characters. A block of eight only bites when eight
+// digits are left, so a run of 4 to 7 digits used to fall back to one digit at a
+// time; taking four of them at once is what makes the tail of a long fraction
+// cheap. Adapted from fast_float, credit @aqrit.
+simdjson_inline bool is_made_of_four_digits_fast(const uint8_t *chars) {
+  uint32_t val;
+  // Reads up to 3 bytes beyond the digits, which SIMDJSON_PADDING covers.
+  static_assert(3 <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be bigger than 3");
+  std::memcpy(&val, chars, 4);
+  return !(((val + 0x46464646) | (val - 0x30303030)) & 0x80808080);
+}
+
+// Only call this when is_made_of_four_digits_fast() says the four characters are
+// digits. Little-endian: guarded by SIMDJSON_SWAR_NUMBER_PARSING.
+simdjson_inline uint32_t parse_four_digits_unrolled(const uint8_t *chars) {
+  uint32_t val;
+  std::memcpy(&val, chars, 4);
+  val -= 0x30303030;
+  val = (val * 10) + (val >> 8);
+  return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
 template<typename I>
 SIMDJSON_NO_SANITIZE_UNDEFINED // We deliberately allow overflow here and check later
 simdjson_inline bool parse_digit(const uint8_t c, I &i) {
@@ -34416,26 +40505,73 @@ simdjson_inline bool is_digit(const uint8_t c) {
   return static_cast<uint8_t>(c - '0') <= 9;
 }

-simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
-  // we continue with the fiction that we have an integer. If the
-  // floating point number is representable as x * 10^z for some integer
-  // z that fits in 53 bits, then we will be able to convert back the
-  // the integer into a float in a lossless manner.
-  const uint8_t *const first_after_period = p;
+// Consumes a run of digits into i, eight at a time while eight are available,
+// then four, then one at a time. Overflow is deliberate: the caller counts the
+// digits and falls back when there are too many for a 64-bit mantissa.
+//
+// Long runs of digits are what a float-heavy document is made of: the fraction
+// of a binary64 printed to full precision is 15 to 17 digits, and reading those
+// one at a time is the single largest cost in parse_double(). Integer parts are
+// left alone: they are usually a handful of digits, and there the check for a
+// block of eight is wasted work.
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_fraction_digits(const uint8_t *&p, uint64_t &i) {
+#ifdef SIMDJSON_SWAR_NUMBER_PARSING
+#if SIMDJSON_SWAR_NUMBER_PARSING
+  while (is_made_of_eight_digits_fast(p)) {
+    i = i * 100000000 + parse_eight_digits_unrolled(p);
+    p += 8;
+  }
+  // A 4 to 7 digit remainder is the common case once the blocks of eight are
+  // gone: 15 fraction digits leave 7, and 17 leave 1 after two blocks.
+  if (is_made_of_four_digits_fast(p)) {
+    i = i * 10000 + parse_four_digits_unrolled(p);
+    p += 4;
+  }
+#endif // SIMDJSON_SWAR_NUMBER_PARSING
+#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
+  while (parse_digit(*p, i)) { p++; }
+}
+
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_float_integer_digits(const uint8_t *&p, uint64_t &i,
+                                                bool &leading_zero) {
+  if (!parse_digit(*p, i)) { leading_zero = true; return; }
+  p++;
+  leading_zero = (i == 0);
+  while (parse_digit(*p, i)) { p++; }
+}

+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_integer_digits(const uint8_t *&p, uint64_t &i) {
 #ifdef SIMDJSON_SWAR_NUMBER_PARSING
 #if SIMDJSON_SWAR_NUMBER_PARSING
-  // this helps if we have lots of decimals!
-  // this turns out to be frequent enough.
+  // Identifiers, timestamps and counters often have eight digits or more.
+  // Prior related work: jsonifier parses integers as eight-digit SWAR words
+  // (str_to_i.hpp, https://github.com/nihilai-collective/Jsonifier). This
+  // takes one such word with parse_eight_digits_unrolled, the routine
+  // simdjson uses for long fractions.
   if (is_made_of_eight_digits_fast(p)) {
     i = i * 100000000 + parse_eight_digits_unrolled(p);
     p += 8;
   }
+  const uint8_t *const swar_end = p + 8;
+  while (p < swar_end && is_made_of_four_digits_fast(p)) {
+    i = i * 10000 + parse_four_digits_unrolled(p);
+    p += 4;
+  }
 #endif // SIMDJSON_SWAR_NUMBER_PARSING
 #endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
-  // Unrolling the first digit makes a small difference on some implementations (e.g. westmere)
-  if (parse_digit(*p, i)) { ++p; }
   while (parse_digit(*p, i)) { p++; }
+}
+
+simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
+  // we continue with the fiction that we have an integer. If the
+  // floating point number is representable as x * 10^z for some integer
+  // z that fits in 53 bits, then we will be able to convert back the
+  // the integer into a float in a lossless manner.
+  const uint8_t *const first_after_period = p;
+  parse_fraction_digits(p, i);
   exponent = first_after_period - p;
   // Decimal without digits (123.) is illegal
   if (exponent == 0) {
@@ -34524,7 +40660,7 @@ simdjson_inline size_t significant_digits(const uint8_t * start_digits, size_t d
 } // unnamed namespace

 /** @private */
-static error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
+inline error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
   if (parse_float_fallback(src, answer)) {
     return SUCCESS;
   }
@@ -34607,15 +40743,17 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
   return SUCCESS;              // always succeeds
 }

-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept  { return false; }
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept  { return false; }
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept { return number_type::signed_integer; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline bool is_negative(const uint8_t *) noexcept  { return false; }
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t *) noexcept  { return false; }
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t *) noexcept { return number_type::signed_integer; }
 #else

 // parse the number at src
@@ -34646,7 +40784,21 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
   size_t digit_count = size_t(p - start_digits);
-  if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) { return INVALID_NUMBER(src); }
+  if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // By this point, we know that our input does not begin with a digit. We will attempt
+    // to handle NaN/Infinity.
+
+    double d;
+    if (compute_nan_inf(p, negative, d)) {
+      writer.append_double(d);
+      return SUCCESS;
+    }
+#endif
+
+    return INVALID_NUMBER(src);
+  }

   //
   // Handle floats if there is a . or e (or both)
@@ -34695,7 +40847,7 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
     // - Therefore, if the number is positive and lower than that, it's overflow.
     // - The value we are looking at is less than or equal to INT64_MAX.
     //
-    }  else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return INVALID_NUMBER(src); }
+    }  else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return BIGINT_NUMBER(src); }
   }

   // Write unsigned if it does not fit in a signed integer.
@@ -34794,7 +40946,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const u
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -34892,7 +41044,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_stri
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -34947,7 +41099,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uin
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -35033,7 +41185,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = src;
   uint64_t i = 0;
-  while (parse_digit(*src, i)) { src++; }
+  parse_integer_digits(src, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -35073,11 +41225,20 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   //
   uint64_t i = 0;
   const uint8_t *p = src;
-  p += parse_digit(*p, i);
-  bool leading_zero = (i == 0);
-  while (parse_digit(*p, i)) { p++; }
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
   // no integer digits, or 0123 (zero must be solo)
-  if ( p == src ) { return INCORRECT_TYPE; }
+  if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no loading digits, the number may be nan or infinity.
+    // Attempt to compute those, and return on success.
+    double d;
+    if (compute_nan_inf(p, negative, d)) { return d; }
+#endif
+
+    return INCORRECT_TYPE;
+  }
   if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }

   //
@@ -35088,9 +41249,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   if (simdjson_likely(*p == '.')) {
     p++;
     const uint8_t *start_decimal_digits = p;
-    if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
-    p++;
-    while (parse_digit(*p, i)) { p++; }
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
     exponent = -(p - start_decimal_digits);

     // Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -35139,6 +41299,97 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   return d;
 }

+// Parse a JSON number into a binary32 (float) value.
+//
+// This mirrors parse_double, but it rounds to binary32 directly instead of
+// rounding to binary64 and then to binary32: the latter is subject to double
+// rounding and would not always produce the nearest float.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * src) noexcept {
+  //
+  // Check for minus sign
+  //
+  bool negative = (*src == '-');
+  src += uint8_t(negative);
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, the number may be nan or infinity.
+    // Attempt to compute those, and return on success.
+    float f;
+    if (compute_nan_inf(p, negative, f)) { return f; }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = p-start_digits > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if (*p == 'e' || *p == 'E') {
+    p++;
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while (parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  float d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_32(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
 simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
   return (*src == '-');
 }
@@ -35291,11 +41542,25 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   //
   uint64_t i = 0;
   const uint8_t *p = src;
-  p += parse_digit(*p, i);
-  bool leading_zero = (i == 0);
-  while (parse_digit(*p, i)) { p++; }
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
   // no integer digits, or 0123 (zero must be solo)
-  if ( p == src ) { return INCORRECT_TYPE; }
+  if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, attempt to parse numbers that are either
+    // NaN or Infinity
+    if (atomparsing::is_valid_inf_in_string(src)) {
+      double inf = std::numeric_limits<double>::infinity();
+      return negative ? -inf : inf;
+    }
+
+    if (atomparsing::is_valid_nan_in_string(src)) {
+      return std::numeric_limits<double>::quiet_NaN();
+    }
+#endif
+
+    return INCORRECT_TYPE;
+  }
   if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }

   //
@@ -35306,9 +41571,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   if (simdjson_likely(*p == '.')) {
     p++;
     const uint8_t *start_decimal_digits = p;
-    if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
-    p++;
-    while (parse_digit(*p, i)) { p++; }
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
     exponent = -(p - start_decimal_digits);

     // Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -35357,6 +41621,99 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   return d;
 }

+// Parse a JSON number held inside a JSON string into a binary32 (float) value.
+// See parse_float for why we do not simply round parse_double_in_string.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * src) noexcept {
+  //
+  // Check for minus sign
+  //
+  bool negative = (*(src + 1) == '-');
+  src += uint8_t(negative) + 1;
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, attempt to parse numbers that are either
+    // NaN or Infinity
+    if (atomparsing::is_valid_inf_in_string(src)) {
+      float inf = std::numeric_limits<float>::infinity();
+      return negative ? -inf : inf;
+    }
+
+    if (atomparsing::is_valid_nan_in_string(src)) {
+      return std::numeric_limits<float>::quiet_NaN();
+    }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = p-start_digits > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if (*p == 'e' || *p == 'E') {
+    p++;
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while (parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if (*p != '"') { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  float d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_32(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
 } // unnamed namespace
 #endif // SIMDJSON_SKIPNUMBERPARSING

@@ -35622,11 +41979,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
   return __builtin_popcountll(input_num);
 }

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
-                                uint64_t *result) {
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-}

 } // unnamed namespace
 } // namespace rvv_vls
@@ -36367,6 +42719,9 @@ namespace atomparsing {
 // to the compile-time constant 1936482662.
 simdjson_inline uint32_t string_to_uint32(const char* str) { uint32_t val; std::memcpy(&val, str, sizeof(uint32_t)); return val; }

+// Acts on the same principle as string_to_uint32, but on an 8-byte block of memory
+simdjson_inline uint64_t string_to_uint64(const char* str) { uint64_t val; std::memcpy(&val, str, sizeof(uint64_t)); return val; }
+

 // Again in str4ncmp we use a memcpy to avoid undefined behavior. The memcpy may appear expensive.
 // Yet all decent optimizing compilers will compile memcpy to a single instruction, just about.
@@ -36378,6 +42733,28 @@ simdjson_inline uint32_t str4ncmp(const uint8_t *src, const char* atom) {
   return srcval ^ string_to_uint32(atom);
 }

+// Checks that the first 8 characters of the input string match the given atom in a case-insensitive manner.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint64_t str8ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+  uint64_t srcval; // we want to avoid unaligned 32-bit loads (undefined in C/C++)
+  static_assert(sizeof(uint64_t) <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be larger than 8 bytes");
+  std::memcpy(&srcval, src, sizeof(uint64_t));
+
+  return (srcval | 0x2020202020202020ull) ^ string_to_uint64(atom);
+}
+
+// Checks that the first 3 characters of 'src' match 'atom' in a case-insensitive way.
+//
+// 'atom' must consist of only lowercase letters.
+simdjson_warn_unused
+simdjson_inline uint32_t str3ncmp_case_insensitive(const uint8_t *src, const char* atom) {
+  return ((src[0] | 0x20) ^ atom[0]) //
+       | ((src[1] | 0x20) ^ atom[1]) //
+       | ((src[2] | 0x20) ^ atom[2]);
+}
+
 simdjson_warn_unused
 simdjson_inline bool is_valid_true_atom(const uint8_t *src) {
   return (str4ncmp(src, "true") | jsoncharutils::is_not_structural_or_whitespace(src[4])) == 0;
@@ -36414,6 +42791,71 @@ simdjson_inline bool is_valid_null_atom(const uint8_t *src, size_t len) {
   else { return false; }
 }

+#if SIMDJSON_ENABLE_NAN_INF
+// "nan" is 3 bytes; we check characters and then verify the next
+// character is structural or whitespace. We accept both "nan" and "NaN".
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src) {
+  return (str3ncmp_case_insensitive(src, "nan")
+        | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+}
+
+// checks that the next four characters of a string are 'nan"', where the 'nan'
+// is checked in a case-insensitive way.
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_in_string(const uint8_t *src) {
+  return (str3ncmp_case_insensitive(src, "nan") | (src[3] ^ '"')) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_nan_atom(const uint8_t *src, size_t len) {
+  if (len > 3) { return is_valid_nan_atom(src); }
+  if (len == 3) { return str3ncmp_case_insensitive(src, "nan") == 0; }
+  return false;
+}
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src) {
+  bool is_short_inf = (str3ncmp_case_insensitive(src, "inf")
+                     | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+  if(is_short_inf) return true;
+
+  // Check for 'infinity' (any capitalization)
+  return (str8ncmp_case_insensitive(src, "infinity") | jsoncharutils::is_not_structural_or_whitespace(src[8])) == 0;
+}
+
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_in_string(const uint8_t *src) {
+  bool is_short_inf = (str3ncmp_case_insensitive(src, "inf") | (src[3] ^ '"')) == 0;
+  if(is_short_inf) return true;
+
+  return (str8ncmp_case_insensitive(src, "infinity") | (src[8] ^ '"')) == 0;
+}
+
+
+// This function will accept any case-insensitive 3-character spelling of
+// infinity: 'inf', 'INF', and 'Inf' are all accepted.
+//
+// Any capitalization of 'infinity' is also accepted.
+simdjson_warn_unused
+simdjson_inline bool is_valid_inf_atom(const uint8_t *src, size_t len) {
+  if (len > 8) { return is_valid_inf_atom(src); }
+  if (len == 8 && str8ncmp_case_insensitive(src, "infinity") == 0) {
+    return true;
+  }
+  if (len > 3) {
+    return (str3ncmp_case_insensitive(src, "inf")
+          | jsoncharutils::is_not_structural_or_whitespace(src[3])) == 0;
+  }
+  if (len == 3) { return str3ncmp_case_insensitive(src, "inf") == 0; }
+  return false;
+}
+#endif // SIMDJSON_ENABLE_NAN_INF
+
 } // namespace atomparsing
 } // unnamed namespace
 } // namespace rvv_vls
@@ -36504,7 +42946,7 @@ inline simdjson_warn_unused error_code dom_parser_implementation::set_capacity(s
 }

 inline simdjson_warn_unused error_code dom_parser_implementation::set_max_depth(size_t max_depth) noexcept {
-  if(max_depth == 0 || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
+  if(max_depth == 0 || max_depth > SIMDJSON_MAX_DEPTH || max_depth > SIZE_MAX / sizeof(open_container)) { return CAPACITY; }
   // Stage 2 stacks
   open_containers.reset(new (std::nothrow) open_container[max_depth]);
   is_array.reset(new (std::nothrow) bool[max_depth]);
@@ -36683,6 +43125,7 @@ protected:
 /* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_NUMBERPARSING_H */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/base.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/jsoncharutils.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/atomparsing.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/internal/numberparsing_tables.h" */
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */

@@ -36722,6 +43165,18 @@ simdjson_inline double to_double(uint64_t mantissa, uint64_t real_exponent, bool
     return d;
 }

+// Convert a mantissa, an exponent and a sign bit into an ieee32 float (binary32).
+// The real_exponent needs to be in [0, 254] (technically real_exponent = 255 would be acceptable).
+// The mantissa should be in [0,1<<24). The bit at index (1U << 23) will be zeroed.
+simdjson_inline float to_float(uint32_t mantissa, uint32_t real_exponent, bool negative) {
+    float f;
+    mantissa &= ~(uint32_t(1) << 23);
+    mantissa |= real_exponent << 23;
+    mantissa |= ((static_cast<uint32_t>(negative)) << 31);
+    std::memcpy(&f, &mantissa, sizeof(f));
+    return f;
+}
+
 // Attempts to compute i * 10^(power) exactly; and if "negative" is
 // true, negate the result.
 // This function will only work in some cases, when it does not work, success is
@@ -36978,6 +43433,219 @@ simdjson_inline bool compute_float_64(int64_t power, uint64_t i, bool negative,
   return true;
 }

+// Attempts to compute i * 10^(power) as a binary32 (float) value; and if
+// "negative" is true, negate the result.
+//
+// This is the same Eisel-Lemire algorithm as compute_float_64, retargeted at
+// binary32. It is adapted from fast_float
+// (https://github.com/fastfloat/fast_float), where the routine is written once
+// and instantiated for each binary format. We only need 24 bits of mantissa
+// instead of 53, so we ask the truncated product for 24 + 2 = 26 accurate bits
+// (plus one bit that may be lost to the "upperbit" shift), and we shift the
+// product right by (upperbit + 64 - 23 - 3) instead of (upperbit + 64 - 52 - 3).
+//
+// The power-of-five table is shared with the binary64 code: the useful range
+// for binary32, [smallest_power_binary32, largest_power_binary32], is a strict
+// subset of [smallest_power, largest_power].
+//
+// The function returns false when the result would be infinite: simdjson
+// refuses to parse infinite values, so the caller reports an error. Unlike
+// compute_float_64, a false return never means "try harder": the accuracy of
+// the two-limb product is guaranteed by Noble Mushtak and Daniel Lemire,
+// "Fast Number Parsing Without Fallback" (https://arxiv.org/abs/2212.06644).
+// The caller must still fall back to a slow path when the 64-bit mantissa i
+// was truncated (more than 19 significant digits).
+//
+// We assume that power is in the [smallest_power, largest_power] interval: the
+// caller is responsible for this check.
+simdjson_inline bool compute_float_32(int64_t power, uint64_t i, bool negative, float &d) {
+  // Powers of ten that are exactly representable as binary32 values: 10^k is
+  // exact as long as k <= 10 (5^10 = 9765625 < 2**24 while 5^11 > 2**24).
+  static constexpr float power_of_ten_float[] = {
+      1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+      1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+  // The range of powers of ten that a non-zero, finite binary32 value can be
+  // built from. Anything smaller rounds to zero, anything larger is infinite.
+  // These are the smallest_power_of_ten()/largest_power_of_ten() constants that
+  // fast_float uses for binary32.
+  constexpr int smallest_power_binary32 = -65;
+  constexpr int largest_power_binary32 = 38;
+
+  // We start with the fast path described in
+  // Clinger WD. How to read floating point numbers accurately.
+  // ACM SIGPLAN Notices. 1990
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+  // We cannot be certain that x/y is rounded to nearest.
+  if (0 <= power && power <= 10 && i <= 16777215)
+#else
+  if (-10 <= power && power <= 10 && i <= 16777215)
+#endif
+  {
+    // Convert the integer into a float. This is lossless since
+    // 0 <= i <= 2^24 - 1.
+    d = float(i);
+    // Both d and the power of ten are exactly representable as binary32
+    // values, so the product (or quotient) is correctly rounded.
+    if (power < 0) {
+      d = d / power_of_ten_float[-power];
+    } else {
+      d = d * power_of_ten_float[power];
+    }
+    if (negative) {
+      d = -d;
+    }
+    return true;
+  }
+
+  // The fast path has failed, so we fall back on the Eisel-Lemire algorithm.
+  // It needs i > 0 (so that the leading bit of i can be normalized), so we
+  // handle i == 0 separately. We also handle the powers of ten that are so
+  // small (or so large) that the answer is zero (or infinite) whatever the
+  // mantissa is.
+  if (i == 0 || power < smallest_power_binary32) {
+    d = negative ? -0.0f : 0.0f;
+    return true;
+  }
+  if (power > largest_power_binary32) {
+    // We have, for sure, an infinite value.
+    return false;
+  }
+
+  // The exponent is 128 + 63 + power + floor(log(5**power)/log(2)).
+  // The 128 comes from the ieee32 standard (the minimal exponent is -127).
+  // The 63 comes from the fact that we use a 64-bit word.
+  // See compute_float_64 for a discussion of the magical 152170 + 65536.
+  int64_t exponent = (((152170 + 65536) * power) >> 16) + 128 + 63;
+
+  // We want the most significant bit of i to be 1. Shift if needed.
+  int lz = leading_zeroes(i);
+  i <<= lz;
+
+  // We want the most significant 64 bits of the product i * 5**power. It is
+  // safe to index the table because
+  // smallest_power <= smallest_power_binary32 <= power
+  //               <= largest_power_binary32 <= largest_power.
+  const uint32_t index = 2 * uint32_t(power - simdjson::internal::smallest_power);
+#if SIMDJSON_STATIC_REFLECTION
+  simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index]);
+#else
+  simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index]);
+#endif
+
+  // Unless the least significant 38 bits of the high (64-bit) part of the full
+  // product are all 1s, then we know that the most significant 26 bits are
+  // exact and no further work is needed. Having 26 bits is necessary because
+  // we need 24 bits for the mantissa but we have to have one rounding bit and
+  // we can waste a bit if the most significant bit of the product is zero.
+  // (For binary64, the same reasoning gives 55 bits and a 9-bit mask.)
+  if((firstproduct.high & 0x3FFFFFFFFF) == 0x3FFFFFFFFF) {
+    // The truncated multiplication was not accurate enough; use the next 64
+    // bits of the power of five to refine it. See compute_float_64 for a
+    // detailed discussion.
+#if SIMDJSON_STATIC_REFLECTION
+    simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index + 1]);
+#else
+    simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index + 1]);
+#endif
+    firstproduct.low += secondproduct.high;
+    if(secondproduct.high > firstproduct.low) { firstproduct.high++; }
+  }
+  uint64_t lower = firstproduct.low;
+  uint64_t upper = firstproduct.high;
+  // The final mantissa should be 24 bits with a leading 1.
+  // We shift it so that it occupies 25 bits with a leading 1.
+  ///////
+  uint64_t upperbit = upper >> 63;
+  uint64_t mantissa = upper >> (upperbit + 38); // 38 == 64 - 23 - 3
+  lz += int(1 ^ upperbit);
+
+  // Here we have mantissa < (1<<25).
+  int64_t real_exponent = exponent - lz;
+  if (simdjson_unlikely(real_exponent <= 0)) { // we have a subnormal?
+    // Here we have that real_exponent <= 0 so -real_exponent >= 0
+    if(-real_exponent + 1 >= 64) { // if we have more than 64 bits below the minimum exponent, you have a zero for sure.
+      d = negative ? -0.0f : 0.0f;
+      return true;
+    }
+    // next line is safe because -real_exponent + 1 < 64
+    mantissa >>= -real_exponent + 1;
+    // Thankfully, we can't have both "round-to-even" and subnormals because
+    // "round-to-even" only occurs for powers close to 0.
+    mantissa += (mantissa & 1); // round up
+    mantissa >>= 1;
+    // As in compute_float_64, rounding up may take us out of the subnormal
+    // range, so we can only decide after rounding.
+    real_exponent = (mantissa < (uint64_t(1) << 23)) ? 0 : 1;
+    d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+    return true;
+  }
+  // We have to round to even. The "to even" part is only a problem when we are
+  // right in between two floats, which we guard against. The bounds on the
+  // power of ten are those of fast_float's
+  // min_exponent_round_to_even()/max_exponent_round_to_even() for binary32:
+  // when q >= 0, (2m+1) must be divisible by 5^q with 5^q <= 2^25, so q <= 10;
+  // when q < 0, we need 2^24 x 5^{-q} < 2^{64}, so q >= -17.
+  if (simdjson_unlikely((lower <= 1) && (power >= -17) && (power <= 10) && ((mantissa & 3) == 1))) {
+    if((mantissa << (upperbit + 38)) == upper) {
+      mantissa &= ~uint64_t(1);   // flip it so that we do not round up
+    }
+  }
+
+  mantissa += mantissa & 1;
+  mantissa >>= 1;
+
+  // Here we have mantissa < (1<<24), unless there was an overflow
+  if (mantissa >= (uint64_t(1) << 24)) {
+    mantissa = (uint64_t(1) << 23);
+    real_exponent++;
+  }
+  mantissa &= ~(uint64_t(1) << 23);
+  // we have to check that real_exponent is in range, otherwise we bail out
+  if (simdjson_unlikely(real_exponent > 254)) {
+    // We have an infinite value!!! We could actually throw an error here if we could.
+    return false;
+  }
+  d = to_float(uint32_t(mantissa), uint32_t(real_exponent), negative);
+  return true;
+}
+
+#if SIMDJSON_ENABLE_NAN_INF
+// Parses a nan or infinity. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, double& d) noexcept {
+  if (atomparsing::is_valid_inf_atom(src)) {
+    double inf = std::numeric_limits<double>::infinity();
+    d = negative ? -inf : inf;
+    return true;
+  }
+
+  if (atomparsing::is_valid_nan_atom(src)) {
+    d = std::numeric_limits<double>::quiet_NaN();
+    return true;
+  }
+
+  return false;
+}
+
+// Parses a nan or infinity as a binary32 value. Returns true on success, false on failure.
+simdjson_unused simdjson_inline bool compute_nan_inf(const uint8_t* src, bool negative, float& d) noexcept {
+  if (atomparsing::is_valid_inf_atom(src)) {
+    float inf = std::numeric_limits<float>::infinity();
+    d = negative ? -inf : inf;
+    return true;
+  }
+
+  if (atomparsing::is_valid_nan_atom(src)) {
+    d = std::numeric_limits<float>::quiet_NaN();
+    return true;
+  }
+
+  return false;
+}
+#endif
+
 // We call a fallback floating-point parser that might be slow. Note
 // it will accept JSON numbers, but the JSON spec. is more restrictive so
 // before you call parse_float_fallback, you need to have validated the input
@@ -37015,6 +43683,16 @@ static bool parse_float_fallback(const uint8_t *ptr, const uint8_t *end_ptr, dou
   return !(*outDouble > (std::numeric_limits<double>::max)() || *outDouble < std::numeric_limits<double>::lowest());
 }

+// Same as parse_float_fallback, but for binary32 (float) values. Going through
+// the binary64 fallback and then rounding to binary32 would be subject to
+// double rounding, so we run the fallback algorithm directly on binary32.
+static bool parse_float_fallback(const uint8_t *ptr, float *outFloat) {
+  *outFloat = simdjson::internal::from_chars_float(reinterpret_cast<const char *>(ptr));
+  // We do not accept infinite values. See the binary64 version above for why we
+  // do not use std::isfinite.
+  return !(*outFloat > (std::numeric_limits<float>::max)() || *outFloat < std::numeric_limits<float>::lowest());
+}
+
 // check quickly whether the next 8 chars are made of digits
 // at a glance, it looks better than Mula's
 // http://0x80.pl/articles/swar-digits-validate.html
@@ -37033,6 +43711,28 @@ simdjson_inline bool is_made_of_eight_digits_fast(const uint8_t *chars) {
           0x3333333333333333);
 }

+// The same idea for four characters. A block of eight only bites when eight
+// digits are left, so a run of 4 to 7 digits used to fall back to one digit at a
+// time; taking four of them at once is what makes the tail of a long fraction
+// cheap. Adapted from fast_float, credit @aqrit.
+simdjson_inline bool is_made_of_four_digits_fast(const uint8_t *chars) {
+  uint32_t val;
+  // Reads up to 3 bytes beyond the digits, which SIMDJSON_PADDING covers.
+  static_assert(3 <= SIMDJSON_PADDING, "SIMDJSON_PADDING must be bigger than 3");
+  std::memcpy(&val, chars, 4);
+  return !(((val + 0x46464646) | (val - 0x30303030)) & 0x80808080);
+}
+
+// Only call this when is_made_of_four_digits_fast() says the four characters are
+// digits. Little-endian: guarded by SIMDJSON_SWAR_NUMBER_PARSING.
+simdjson_inline uint32_t parse_four_digits_unrolled(const uint8_t *chars) {
+  uint32_t val;
+  std::memcpy(&val, chars, 4);
+  val -= 0x30303030;
+  val = (val * 10) + (val >> 8);
+  return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
 template<typename I>
 SIMDJSON_NO_SANITIZE_UNDEFINED // We deliberately allow overflow here and check later
 simdjson_inline bool parse_digit(const uint8_t c, I &i) {
@@ -37049,26 +43749,73 @@ simdjson_inline bool is_digit(const uint8_t c) {
   return static_cast<uint8_t>(c - '0') <= 9;
 }

-simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
-  // we continue with the fiction that we have an integer. If the
-  // floating point number is representable as x * 10^z for some integer
-  // z that fits in 53 bits, then we will be able to convert back the
-  // the integer into a float in a lossless manner.
-  const uint8_t *const first_after_period = p;
+// Consumes a run of digits into i, eight at a time while eight are available,
+// then four, then one at a time. Overflow is deliberate: the caller counts the
+// digits and falls back when there are too many for a 64-bit mantissa.
+//
+// Long runs of digits are what a float-heavy document is made of: the fraction
+// of a binary64 printed to full precision is 15 to 17 digits, and reading those
+// one at a time is the single largest cost in parse_double(). Integer parts are
+// left alone: they are usually a handful of digits, and there the check for a
+// block of eight is wasted work.
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_fraction_digits(const uint8_t *&p, uint64_t &i) {
+#ifdef SIMDJSON_SWAR_NUMBER_PARSING
+#if SIMDJSON_SWAR_NUMBER_PARSING
+  while (is_made_of_eight_digits_fast(p)) {
+    i = i * 100000000 + parse_eight_digits_unrolled(p);
+    p += 8;
+  }
+  // A 4 to 7 digit remainder is the common case once the blocks of eight are
+  // gone: 15 fraction digits leave 7, and 17 leave 1 after two blocks.
+  if (is_made_of_four_digits_fast(p)) {
+    i = i * 10000 + parse_four_digits_unrolled(p);
+    p += 4;
+  }
+#endif // SIMDJSON_SWAR_NUMBER_PARSING
+#endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
+  while (parse_digit(*p, i)) { p++; }
+}
+
+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_float_integer_digits(const uint8_t *&p, uint64_t &i,
+                                                bool &leading_zero) {
+  if (!parse_digit(*p, i)) { leading_zero = true; return; }
+  p++;
+  leading_zero = (i == 0);
+  while (parse_digit(*p, i)) { p++; }
+}

+SIMDJSON_NO_SANITIZE_UNDEFINED
+simdjson_inline void parse_integer_digits(const uint8_t *&p, uint64_t &i) {
 #ifdef SIMDJSON_SWAR_NUMBER_PARSING
 #if SIMDJSON_SWAR_NUMBER_PARSING
-  // this helps if we have lots of decimals!
-  // this turns out to be frequent enough.
+  // Identifiers, timestamps and counters often have eight digits or more.
+  // Prior related work: jsonifier parses integers as eight-digit SWAR words
+  // (str_to_i.hpp, https://github.com/nihilai-collective/Jsonifier). This
+  // takes one such word with parse_eight_digits_unrolled, the routine
+  // simdjson uses for long fractions.
   if (is_made_of_eight_digits_fast(p)) {
     i = i * 100000000 + parse_eight_digits_unrolled(p);
     p += 8;
   }
+  const uint8_t *const swar_end = p + 8;
+  while (p < swar_end && is_made_of_four_digits_fast(p)) {
+    i = i * 10000 + parse_four_digits_unrolled(p);
+    p += 4;
+  }
 #endif // SIMDJSON_SWAR_NUMBER_PARSING
 #endif // #ifdef SIMDJSON_SWAR_NUMBER_PARSING
-  // Unrolling the first digit makes a small difference on some implementations (e.g. westmere)
-  if (parse_digit(*p, i)) { ++p; }
   while (parse_digit(*p, i)) { p++; }
+}
+
+simdjson_warn_unused simdjson_inline error_code parse_decimal_after_separator(simdjson_unused const uint8_t *const src, const uint8_t *&p, uint64_t &i, int64_t &exponent) {
+  // we continue with the fiction that we have an integer. If the
+  // floating point number is representable as x * 10^z for some integer
+  // z that fits in 53 bits, then we will be able to convert back the
+  // the integer into a float in a lossless manner.
+  const uint8_t *const first_after_period = p;
+  parse_fraction_digits(p, i);
   exponent = first_after_period - p;
   // Decimal without digits (123.) is illegal
   if (exponent == 0) {
@@ -37157,7 +43904,7 @@ simdjson_inline size_t significant_digits(const uint8_t * start_digits, size_t d
 } // unnamed namespace

 /** @private */
-static error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
+inline error_code slow_float_parsing(simdjson_unused const uint8_t * src, double* answer) {
   if (parse_float_fallback(src, answer)) {
     return SUCCESS;
   }
@@ -37240,15 +43987,17 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
   return SUCCESS;              // always succeeds
 }

-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const src) noexcept { return 0; }
-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept  { return false; }
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept  { return false; }
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept { return number_type::signed_integer; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * const) noexcept { return 0; }
+simdjson_unused simdjson_inline bool is_negative(const uint8_t *) noexcept  { return false; }
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t *) noexcept  { return false; }
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t *) noexcept { return number_type::signed_integer; }
 #else

 // parse the number at src
@@ -37279,7 +44028,21 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
   size_t digit_count = size_t(p - start_digits);
-  if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) { return INVALID_NUMBER(src); }
+  if (digit_count == 0 || ('0' == *start_digits && digit_count > 1)) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // By this point, we know that our input does not begin with a digit. We will attempt
+    // to handle NaN/Infinity.
+
+    double d;
+    if (compute_nan_inf(p, negative, d)) {
+      writer.append_double(d);
+      return SUCCESS;
+    }
+#endif
+
+    return INVALID_NUMBER(src);
+  }

   //
   // Handle floats if there is a . or e (or both)
@@ -37328,7 +44091,7 @@ simdjson_warn_unused simdjson_inline error_code parse_number(const uint8_t *cons
     // - Therefore, if the number is positive and lower than that, it's overflow.
     // - The value we are looking at is less than or equal to INT64_MAX.
     //
-    }  else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return INVALID_NUMBER(src); }
+    }  else if (src[0] != uint8_t('1') || i <= uint64_t(INT64_MAX)) { return BIGINT_NUMBER(src); }
   }

   // Write unsigned if it does not fit in a signed integer.
@@ -37427,7 +44190,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned(const u
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -37525,7 +44288,7 @@ simdjson_unused simdjson_inline simdjson_result<uint64_t> parse_unsigned_in_stri
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -37580,7 +44343,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer(const uin
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = p;
   uint64_t i = 0;
-  while (parse_digit(*p, i)) { p++; }
+  parse_integer_digits(p, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -37666,7 +44429,7 @@ simdjson_unused simdjson_inline simdjson_result<int64_t> parse_integer_in_string
   // PERF NOTE: we don't use is_made_of_eight_digits_fast because large integers like 123456789 are rare
   const uint8_t *const start_digits = src;
   uint64_t i = 0;
-  while (parse_digit(*src, i)) { src++; }
+  parse_integer_digits(src, i);

   // If there were no digits, or if the integer starts with 0 and has more than one digit, it's an error.
   // Optimization note: size_t is expected to be unsigned.
@@ -37706,11 +44469,20 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   //
   uint64_t i = 0;
   const uint8_t *p = src;
-  p += parse_digit(*p, i);
-  bool leading_zero = (i == 0);
-  while (parse_digit(*p, i)) { p++; }
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
   // no integer digits, or 0123 (zero must be solo)
-  if ( p == src ) { return INCORRECT_TYPE; }
+  if ( p == src ) {
+
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no loading digits, the number may be nan or infinity.
+    // Attempt to compute those, and return on success.
+    double d;
+    if (compute_nan_inf(p, negative, d)) { return d; }
+#endif
+
+    return INCORRECT_TYPE;
+  }
   if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }

   //
@@ -37721,9 +44493,8 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   if (simdjson_likely(*p == '.')) {
     p++;
     const uint8_t *start_decimal_digits = p;
-    if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
-    p++;
-    while (parse_digit(*p, i)) { p++; }
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
     exponent = -(p - start_decimal_digits);

     // Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -37772,67 +44543,12 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   return d;
 }

-simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
-  return (*src == '-');
-}
-
-simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept {
-  bool negative = (*src == '-');
-  src += uint8_t(negative);
-  const uint8_t *p = src;
-  while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
-  if ( p == src ) { return NUMBER_ERROR; }
-  if (jsoncharutils::is_structural_or_whitespace(*p)) { return true; }
-  return false;
-}
-
-simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept {
-  bool negative = (*src == '-');
-  src += uint8_t(negative);
-  const uint8_t *p = src;
-  while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
-  size_t digit_count = size_t(p - src);
-  if ( p == src ) { return NUMBER_ERROR; }
-  if (jsoncharutils::is_structural_or_whitespace(*p)) {
-    static const uint8_t * smaller_big_integer = reinterpret_cast<const uint8_t *>("9223372036854775808");
-    // We have an integer.
-    if(simdjson_unlikely(digit_count > 20)) {
-      return number_type::big_integer;
-    }
-    // If the number is negative and valid, it must be a signed integer.
-    if(negative) {
-      if (simdjson_unlikely(digit_count > 19)) return number_type::big_integer;
-      if (simdjson_unlikely(digit_count == 19 && memcmp(src, smaller_big_integer, 19) > 0)) {
-        return number_type::big_integer;
-      }
-#if SIMDJSON_MINUS_ZERO_AS_FLOAT
-      if(digit_count == 1 && src[0] == '0') {
-        // We have to write -0.0 instead of 0
-        return number_type::floating_point_number;
-      }
-#endif
-      return number_type::signed_integer;
-    }
-    // Let us check if we have a big integer (>=2**64).
-    static const uint8_t * two_to_sixtyfour = reinterpret_cast<const uint8_t *>("18446744073709551616");
-    if((digit_count > 20) || (digit_count == 20 && memcmp(src, two_to_sixtyfour, 20) >= 0)) {
-      return number_type::big_integer;
-    }
-    // The number is positive and smaller than 18446744073709551616 (or 2**64).
-    // We want values larger or equal to 9223372036854775808 to be unsigned
-    // integers, and the other values to be signed integers.
-    if((digit_count == 20) || (digit_count >= 19 && memcmp(src, smaller_big_integer, 19) >= 0)) {
-      return number_type::unsigned_integer;
-    }
-    return number_type::signed_integer;
-  }
-  // Hopefully, we have 'e' or 'E' or '.'.
-  return number_type::floating_point_number;
-}
-
-// Never read at src_end or beyond
-simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * src, const uint8_t * const src_end) noexcept {
-  if(src == src_end) { return NUMBER_ERROR; }
+// Parse a JSON number into a binary32 (float) value.
+//
+// This mirrors parse_double, but it rounds to binary32 directly instead of
+// rounding to binary64 and then to binary32: the latter is subject to double
+// rounding and would not always produce the nearest float.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float(const uint8_t * src) noexcept {
   //
   // Check for minus sign
   //
@@ -37844,91 +44560,20 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8
   //
   uint64_t i = 0;
   const uint8_t *p = src;
-  if(p == src_end) { return NUMBER_ERROR; }
-  p += parse_digit(*p, i);
-  bool leading_zero = (i == 0);
-  while ((p != src_end) && parse_digit(*p, i)) { p++; }
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
   // no integer digits, or 0123 (zero must be solo)
-  if ( p == src ) { return INCORRECT_TYPE; }
-  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+  if ( p == src ) {

-  //
-  // Parse the decimal part.
-  //
-  int64_t exponent = 0;
-  bool overflow;
-  if (simdjson_likely((p != src_end) && (*p == '.'))) {
-    p++;
-    const uint8_t *start_decimal_digits = p;
-    if ((p == src_end) || !parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
-    p++;
-    while ((p != src_end) && parse_digit(*p, i)) { p++; }
-    exponent = -(p - start_decimal_digits);
-
-    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
-    overflow = p-src-1 > 19;
-    if (simdjson_unlikely(overflow && leading_zero)) {
-      // Skip leading 0.00000 and see if it still overflows
-      const uint8_t *start_digits = src + 2;
-      while (*start_digits == '0') { start_digits++; }
-      overflow = start_digits-src > 19;
-    }
-  } else {
-    overflow = p-src > 19;
-  }
-
-  //
-  // Parse the exponent
-  //
-  if ((p != src_end) && (*p == 'e' || *p == 'E')) {
-    p++;
-    if(p == src_end) { return NUMBER_ERROR; }
-    bool exp_neg = *p == '-';
-    p += exp_neg || *p == '+';
-
-    uint64_t exp = 0;
-    const uint8_t *start_exp_digits = p;
-    while ((p != src_end) && parse_digit(*p, exp)) { p++; }
-    // no exp digits, or 20+ exp digits
-    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
-
-    exponent += exp_neg ? 0-exp : exp;
-  }
-
-  if ((p != src_end) && jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
-
-  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, the number may be nan or infinity.
+    // Attempt to compute those, and return on success.
+    float f;
+    if (compute_nan_inf(p, negative, f)) { return f; }
+#endif

-  //
-  // Assemble (or slow-parse) the float
-  //
-  double d;
-  if (simdjson_likely(!overflow)) {
-    if (compute_float_64(exponent, i, negative, d)) { return d; }
-  }
-  if (!parse_float_fallback(src - uint8_t(negative), src_end, &d)) {
-    return NUMBER_ERROR;
+    return INCORRECT_TYPE;
   }
-  return d;
-}
-
-simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * src) noexcept {
-  //
-  // Check for minus sign
-  //
-  bool negative = (*(src + 1) == '-');
-  src += uint8_t(negative) + 1;
-
-  //
-  // Parse the integer part.
-  //
-  uint64_t i = 0;
-  const uint8_t *p = src;
-  p += parse_digit(*p, i);
-  bool leading_zero = (i == 0);
-  while (parse_digit(*p, i)) { p++; }
-  // no integer digits, or 0123 (zero must be solo)
-  if ( p == src ) { return INCORRECT_TYPE; }
   if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }

   //
@@ -37939,9 +44584,239 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   if (simdjson_likely(*p == '.')) {
     p++;
     const uint8_t *start_decimal_digits = p;
-    if (!parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
-    p++;
-    while (parse_digit(*p, i)) { p++; }
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = p-start_digits > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if (*p == 'e' || *p == 'E') {
+    p++;
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while (parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if (jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  float d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_32(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
+simdjson_unused simdjson_inline bool is_negative(const uint8_t * src) noexcept {
+  return (*src == '-');
+}
+
+simdjson_unused simdjson_inline simdjson_result<bool> is_integer(const uint8_t * src) noexcept {
+  bool negative = (*src == '-');
+  src += uint8_t(negative);
+  const uint8_t *p = src;
+  while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
+  if ( p == src ) { return NUMBER_ERROR; }
+  if (jsoncharutils::is_structural_or_whitespace(*p)) { return true; }
+  return false;
+}
+
+simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(const uint8_t * src) noexcept {
+  bool negative = (*src == '-');
+  src += uint8_t(negative);
+  const uint8_t *p = src;
+  while(static_cast<uint8_t>(*p - '0') <= 9) { p++; }
+  size_t digit_count = size_t(p - src);
+  if ( p == src ) { return NUMBER_ERROR; }
+  if (jsoncharutils::is_structural_or_whitespace(*p)) {
+    static const uint8_t * smaller_big_integer = reinterpret_cast<const uint8_t *>("9223372036854775808");
+    // We have an integer.
+    if(simdjson_unlikely(digit_count > 20)) {
+      return number_type::big_integer;
+    }
+    // If the number is negative and valid, it must be a signed integer.
+    if(negative) {
+      if (simdjson_unlikely(digit_count > 19)) return number_type::big_integer;
+      if (simdjson_unlikely(digit_count == 19 && memcmp(src, smaller_big_integer, 19) > 0)) {
+        return number_type::big_integer;
+      }
+#if SIMDJSON_MINUS_ZERO_AS_FLOAT
+      if(digit_count == 1 && src[0] == '0') {
+        // We have to write -0.0 instead of 0
+        return number_type::floating_point_number;
+      }
+#endif
+      return number_type::signed_integer;
+    }
+    // Let us check if we have a big integer (>=2**64).
+    static const uint8_t * two_to_sixtyfour = reinterpret_cast<const uint8_t *>("18446744073709551616");
+    if((digit_count > 20) || (digit_count == 20 && memcmp(src, two_to_sixtyfour, 20) >= 0)) {
+      return number_type::big_integer;
+    }
+    // The number is positive and smaller than 18446744073709551616 (or 2**64).
+    // We want values larger or equal to 9223372036854775808 to be unsigned
+    // integers, and the other values to be signed integers.
+    if((digit_count == 20) || (digit_count >= 19 && memcmp(src, smaller_big_integer, 19) >= 0)) {
+      return number_type::unsigned_integer;
+    }
+    return number_type::signed_integer;
+  }
+  // Hopefully, we have 'e' or 'E' or '.'.
+  return number_type::floating_point_number;
+}
+
+// Never read at src_end or beyond
+simdjson_unused simdjson_inline simdjson_result<double> parse_double(const uint8_t * src, const uint8_t * const src_end) noexcept {
+  if(src == src_end) { return NUMBER_ERROR; }
+  //
+  // Check for minus sign
+  //
+  bool negative = (*src == '-');
+  src += uint8_t(negative);
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  if(p == src_end) { return NUMBER_ERROR; }
+  p += parse_digit(*p, i);
+  bool leading_zero = (i == 0);
+  while ((p != src_end) && parse_digit(*p, i)) { p++; }
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) { return INCORRECT_TYPE; }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely((p != src_end) && (*p == '.'))) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    if ((p == src_end) || !parse_digit(*p, i)) { return NUMBER_ERROR; } // no decimal digits
+    p++;
+    while ((p != src_end) && parse_digit(*p, i)) { p++; }
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = start_digits-src > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if ((p != src_end) && (*p == 'e' || *p == 'E')) {
+    p++;
+    if(p == src_end) { return NUMBER_ERROR; }
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while ((p != src_end) && parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if ((p != src_end) && jsoncharutils::is_not_structural_or_whitespace(*p)) { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  double d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_64(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), src_end, &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
+simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(const uint8_t * src) noexcept {
+  //
+  // Check for minus sign
+  //
+  bool negative = (*(src + 1) == '-');
+  src += uint8_t(negative) + 1;
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, attempt to parse numbers that are either
+    // NaN or Infinity
+    if (atomparsing::is_valid_inf_in_string(src)) {
+      double inf = std::numeric_limits<double>::infinity();
+      return negative ? -inf : inf;
+    }
+
+    if (atomparsing::is_valid_nan_in_string(src)) {
+      return std::numeric_limits<double>::quiet_NaN();
+    }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
     exponent = -(p - start_decimal_digits);

     // Overflow check. More than 19 digits (minus the decimal) may be overflow.
@@ -37990,6 +44865,99 @@ simdjson_unused simdjson_inline simdjson_result<double> parse_double_in_string(c
   return d;
 }

+// Parse a JSON number held inside a JSON string into a binary32 (float) value.
+// See parse_float for why we do not simply round parse_double_in_string.
+simdjson_unused simdjson_inline simdjson_result<float> parse_float_in_string(const uint8_t * src) noexcept {
+  //
+  // Check for minus sign
+  //
+  bool negative = (*(src + 1) == '-');
+  src += uint8_t(negative) + 1;
+
+  //
+  // Parse the integer part.
+  //
+  uint64_t i = 0;
+  const uint8_t *p = src;
+  bool leading_zero;
+  parse_float_integer_digits(p, i, leading_zero);
+  // no integer digits, or 0123 (zero must be solo)
+  if ( p == src ) {
+#if SIMDJSON_ENABLE_NAN_INF
+    // If there are no leading digits, attempt to parse numbers that are either
+    // NaN or Infinity
+    if (atomparsing::is_valid_inf_in_string(src)) {
+      float inf = std::numeric_limits<float>::infinity();
+      return negative ? -inf : inf;
+    }
+
+    if (atomparsing::is_valid_nan_in_string(src)) {
+      return std::numeric_limits<float>::quiet_NaN();
+    }
+#endif
+
+    return INCORRECT_TYPE;
+  }
+  if ( (leading_zero && p != src+1)) { return NUMBER_ERROR; }
+
+  //
+  // Parse the decimal part.
+  //
+  int64_t exponent = 0;
+  bool overflow;
+  if (simdjson_likely(*p == '.')) {
+    p++;
+    const uint8_t *start_decimal_digits = p;
+    parse_fraction_digits(p, i);
+    if (p == start_decimal_digits) { return NUMBER_ERROR; } // no decimal digits
+    exponent = -(p - start_decimal_digits);
+
+    // Overflow check. More than 19 digits (minus the decimal) may be overflow.
+    overflow = p-src-1 > 19;
+    if (simdjson_unlikely(overflow && leading_zero)) {
+      // Skip leading 0.00000 and see if it still overflows
+      const uint8_t *start_digits = src + 2;
+      while (*start_digits == '0') { start_digits++; }
+      overflow = p-start_digits > 19;
+    }
+  } else {
+    overflow = p-src > 19;
+  }
+
+  //
+  // Parse the exponent
+  //
+  if (*p == 'e' || *p == 'E') {
+    p++;
+    bool exp_neg = *p == '-';
+    p += exp_neg || *p == '+';
+
+    uint64_t exp = 0;
+    const uint8_t *start_exp_digits = p;
+    while (parse_digit(*p, exp)) { p++; }
+    // no exp digits, or 20+ exp digits
+    if (p-start_exp_digits == 0 || p-start_exp_digits > 19) { return NUMBER_ERROR; }
+
+    exponent += exp_neg ? 0-exp : exp;
+  }
+
+  if (*p != '"') { return NUMBER_ERROR; }
+
+  overflow = overflow || exponent < simdjson::internal::smallest_power || exponent > simdjson::internal::largest_power;
+
+  //
+  // Assemble (or slow-parse) the float
+  //
+  float d;
+  if (simdjson_likely(!overflow)) {
+    if (compute_float_32(exponent, i, negative, d)) { return d; }
+  }
+  if (!parse_float_fallback(src - uint8_t(negative), &d)) {
+    return NUMBER_ERROR;
+  }
+  return d;
+}
+
 } // unnamed namespace
 #endif // SIMDJSON_SKIPNUMBERPARSING

@@ -38172,6 +45140,376 @@ simdjson_inline implementation_simdjson_result_base<T>::implementation_simdjson_
 // Otherwise, amalgamation will fail.
 /* skipped duplicate #include "simdjson/concepts.h" */
 /* skipped duplicate #include "simdjson/dom/fractured_json.h" */
+/* including simdjson/annotations.h: #include "simdjson/annotations.h" */
+/* begin file simdjson/annotations.h */
+#ifndef SIMDJSON_ANNOTATIONS_H
+#define SIMDJSON_ANNOTATIONS_H
+
+/**
+ * @file annotations.h
+ * @brief Provides compile-time annotations for simdjson structures.
+ * This header defines annotations that can be applied to data members of structures
+ * (and to the structures and enumerations themselves) to control how they are
+ * serialized/deserialized with simdjson. The set of annotations is modelled after
+ * the attributes of the Rust serde library.
+ *
+ * Member annotations:
+ *
+ *   [[= simdjson::rename<"name">]]             use "name" as the JSON key
+ *   [[= simdjson::alias<"a", "b">]]            also accept "a" and "b" when deserializing
+ *   [[= simdjson::skip]]                       never serialize nor deserialize
+ *   [[= simdjson::skip_serializing]]           never serialize
+ *   [[= simdjson::skip_deserializing]]         never deserialize (keeps its current value)
+ *   [[= simdjson::skip_serializing_if<pred>]]  do not serialize when pred(value) is true
+ *   [[= simdjson::default_value]]              a missing key is not an error
+ *   [[= simdjson::default_from<factory>]]      a missing key sets the member to factory()
+ *   [[= simdjson::with<Adapter>]]              custom (de)serialization via Adapter
+ *   [[= simdjson::flatten]]                    inline the members of a nested structure
+ *
+ * Structure (container) annotations:
+ *
+ *   [[= simdjson::rename_all<simdjson::case_style::camel_case>]]  rename every member
+ *   [[= simdjson::default_value]]              no missing key is an error
+ *   [[= simdjson::deny_unknown_fields]]        unknown keys are a deserialization error
+ *   [[= simdjson::transparent]]                (de)serialize as the single member
+ *
+ * Enumeration annotations: rename_all on the enumeration, rename and alias on the
+ * enumerators (e.g., `enum class color { red [[= simdjson::rename<"RED">]] };`).
+ *
+ * This is currently experimental and subject to change (syntax and semantics may evolve).
+ */
+
+#if SIMDJSON_STATIC_REFLECTION
+
+#include <meta>
+#include <string>
+#include <string_view>
+#include <vector>
+
+namespace simdjson {
+
+// Structural compile-time string -- char array avoids the pointer-based
+// 'reflect_constant failed' that occurs with const char* / string_view members.
+template <size_t N>
+struct fixed_string {
+    char data[N];
+
+    consteval fixed_string(const char (&s)[N]) noexcept {
+        for (size_t i = 0; i < N; ++i) { data[i] = s[i]; }
+    }
+
+    consteval std::string_view view() const noexcept { return {data, N - 1}; }
+
+    consteval bool operator==(const fixed_string&) const noexcept = default;
+};
+
+/**
+ * Naming conventions for simdjson::rename_all, mirroring serde's rename_all.
+ * Except for lowercase and uppercase, the C++ identifier is first split into
+ * words at underscores and at case changes ("userId", "user_id" and "UserId" all
+ * give the words "user" and "id"; "HTTPServer" gives "HTTP" and "Server"), and
+ * the words are then joined according to the convention.
+ */
+enum class case_style {
+  lowercase,            ///< every letter lowercased, nothing else changes: userId -> userid
+  uppercase,            ///< every letter uppercased, nothing else changes: user_id -> USER_ID
+  pascal_case,          ///< UserId
+  camel_case,           ///< userId
+  snake_case,           ///< user_id
+  screaming_snake_case, ///< USER_ID
+  kebab_case,           ///< user-id
+  screaming_kebab_case  ///< USER-ID
+};
+
+namespace detail {
+    template <fixed_string Name>
+    struct rename_t {
+        static constexpr auto name = Name;
+        // Exposed as a pointer and a size: std::meta::extract requires structural types.
+        static constexpr const char *key_data = Name.data;
+        static constexpr size_t key_size = Name.view().size();
+    };
+    template <fixed_string... Names>
+    struct alias_t {
+        static_assert(sizeof...(Names) > 0, "simdjson::alias requires at least one name");
+        static constexpr std::string_view keys[] = {Names.view()...};
+        static constexpr const std::string_view *keys_data = keys;
+        static constexpr size_t keys_count = sizeof...(Names);
+    };
+    struct skip_tag {};
+    struct skip_serializing_tag {};
+    struct skip_deserializing_tag {};
+    template <auto Predicate>
+    struct skip_serializing_if_t {
+        static constexpr auto predicate = Predicate;
+    };
+    struct default_value_tag {};
+    template <auto Factory>
+    struct default_from_t {
+        static constexpr auto factory = Factory;
+    };
+    template <typename Adapter>
+    struct with_t {
+        using adapter = Adapter;
+    };
+    template <case_style Style>
+    struct rename_all_t {
+        static constexpr case_style style = Style;
+    };
+    struct deny_unknown_fields_tag {};
+    struct transparent_tag {};
+    struct flatten_tag {};
+
+    // Predicates usable with skip_serializing_if.
+    struct is_none_t {
+        template <typename T>
+        constexpr bool operator()(const T& v) const noexcept { return !v; }
+    };
+    struct is_empty_t {
+        template <typename T>
+        constexpr bool operator()(const T& v) const noexcept { return v.empty(); }
+    };
+} // namespace detail
+
+// Usage: [[= simdjson::rename<"first_name">]] std::string firstName;
+template <fixed_string Name>
+inline constexpr detail::rename_t<Name> rename{};
+
+// Usage: [[= simdjson::alias<"userName", "login">]] std::string user_name;
+// The aliases are accepted (in addition to the regular key) when deserializing.
+// Serialization always uses the regular key. If the JSON object contains more
+// than one of the names, which one is used is unspecified.
+template <fixed_string... Names>
+inline constexpr detail::alias_t<Names...> alias{};
+
+// Usage: [[= simdjson::skip]] int internalCache;
+inline constexpr detail::skip_tag skip{};
+
+// Usage: [[= simdjson::skip_serializing]] std::string password;
+inline constexpr detail::skip_serializing_tag skip_serializing{};
+
+// Usage: [[= simdjson::skip_deserializing]] int computed;
+// The member is never assigned during deserialization (it keeps its current
+// value) and a matching key in the JSON input is treated as unknown.
+inline constexpr detail::skip_deserializing_tag skip_deserializing{};
+
+// Usage: [[= simdjson::skip_serializing_if<simdjson::is_none>]] std::optional<int> x;
+// The predicate is called with the member value; when it returns true, the key
+// is omitted from the output.
+template <auto Predicate>
+inline constexpr detail::skip_serializing_if_t<Predicate> skip_serializing_if{};
+
+// Predicate: true for an empty std::optional, a null smart pointer, etc.
+inline constexpr detail::is_none_t is_none{};
+// Predicate: true for an empty string or container.
+inline constexpr detail::is_empty_t is_empty{};
+
+// Usage: [[= simdjson::default_value]] int port = 8080;
+// When the key is missing from the JSON input, the member is left untouched
+// (with get<T>(), it keeps its default member initializer) instead of reporting
+// NO_SUCH_FIELD. Applied to a structure, it applies to all of its members.
+inline constexpr detail::default_value_tag default_value{};
+
+// Usage: [[= simdjson::default_from<make_port>]] int port;
+// When the key is missing from the JSON input, the member is assigned the
+// result of calling the factory (a constexpr callable taking no argument, such
+// as a captureless lambda or a pointer to a function).
+template <auto Factory>
+inline constexpr detail::default_from_t<Factory> default_from{};
+
+// Usage: [[= simdjson::with<unix_time>]] std::chrono::system_clock::time_point t;
+// Adapter is a type that provides one or both of
+//   static void serialize(simdjson::builder::string_builder &b, const T &value);
+//   static simdjson::error_code deserialize(simdjson::ondemand::value &v, T &out);
+// (the parameters may also be declared auto&). When one of them is missing, the
+// default behaviour is used in that direction.
+template <typename Adapter>
+inline constexpr detail::with_t<Adapter> with{};
+
+// Usage: [[= simdjson::flatten]] pagination page;
+// The members of the nested structure are (de)serialized as if they were members
+// of the enclosing structure: {"id":1,"limit":10,"offset":0} rather than
+// {"id":1,"page":{"limit":10,"offset":0}}. The nested structure's own annotations
+// (rename_all, default_value, ...) apply to its members.
+inline constexpr detail::flatten_tag flatten{};
+
+// Usage: struct [[= simdjson::rename_all<simdjson::case_style::camel_case>]] S {...};
+// Also applies to enumerations. An explicit rename on a member takes precedence.
+template <case_style Style>
+inline constexpr detail::rename_all_t<Style> rename_all{};
+
+// Usage: struct [[= simdjson::deny_unknown_fields]] S {...};
+// Deserialization fails with UNKNOWN_FIELD when the JSON object has a key that
+// is not deserialized into a member (including the keys of skipped members).
+inline constexpr detail::deny_unknown_fields_tag deny_unknown_fields{};
+
+// Usage: struct [[= simdjson::transparent]] user_id { int64_t value; };
+// A structure with a single data member is (de)serialized as that member alone:
+// user_id{42} becomes 42 rather than {"value":42}.
+inline constexpr detail::transparent_tag transparent{};
+
+namespace detail {
+
+// True when the entity (a data member, an enumerator or a type) carries an
+// annotation of type tag.
+consteval bool has_annotation(std::meta::info entity, std::meta::info tag) {
+  return !std::meta::annotations_of_with_type(entity, tag).empty();
+}
+
+// Returns the type of the (first) annotation of entity that is a specialization
+// of the class template tmpl, or std::meta::info{} when there is none.
+consteval std::meta::info annotation_of_template(std::meta::info entity, std::meta::info tmpl) {
+  for (std::meta::info ann : std::meta::annotations_of(entity)) {
+    std::meta::info type = std::meta::type_of(ann);
+    if (std::meta::has_template_arguments(type) && std::meta::template_of(type) == tmpl) {
+      return type;
+    }
+  }
+  return std::meta::info{};
+}
+
+// Value of the static data member `name` of the class `type`.
+template <typename T>
+consteval T static_member_value(std::meta::info type, std::string_view name) {
+  for (std::meta::info m : std::meta::static_data_members_of(type, std::meta::access_context::unchecked())) {
+    if (std::meta::has_identifier(m) && std::meta::identifier_of(m) == name) {
+      return std::meta::extract<T>(m);
+    }
+  }
+  return T{};
+}
+
+consteval bool is_upper(char c) { return c >= 'A' && c <= 'Z'; }
+consteval bool is_lower(char c) { return c >= 'a' && c <= 'z'; }
+consteval bool is_digit(char c) { return c >= '0' && c <= '9'; }
+consteval char to_upper(char c) { return is_lower(c) ? char(c - 'a' + 'A') : c; }
+consteval char to_lower(char c) { return is_upper(c) ? char(c - 'A' + 'a') : c; }
+
+// Split an identifier into words: at underscores, at a lowercase letter or digit
+// followed by an uppercase letter (userId), and before the last capital of an
+// acronym followed by a lowercase letter (HTTPServer -> HTTP, Server).
+consteval std::vector<std::string> split_identifier(std::string_view id) {
+  std::vector<std::string> words;
+  std::string current;
+  for (size_t i = 0; i < id.size(); i++) {
+    char c = id[i];
+    if (c == '_') {
+      if (!current.empty()) { words.push_back(current); current.clear(); }
+      continue;
+    }
+    if (is_upper(c) && !current.empty()) {
+      char prev = current.back();
+      bool next_is_lower = (i + 1 < id.size()) && is_lower(id[i + 1]);
+      if (is_lower(prev) || is_digit(prev) || (is_upper(prev) && next_is_lower)) {
+        words.push_back(current);
+        current.clear();
+      }
+    }
+    current.push_back(c);
+  }
+  if (!current.empty()) { words.push_back(current); }
+  return words;
+}
+
+consteval std::string apply_case_style(std::string_view id, case_style style) {
+  std::string result;
+  if (style == case_style::lowercase || style == case_style::uppercase) {
+    for (char c : id) {
+      result.push_back(style == case_style::lowercase ? to_lower(c) : to_upper(c));
+    }
+    return result;
+  }
+  std::vector<std::string> words = split_identifier(id);
+  for (size_t w = 0; w < words.size(); w++) {
+    const std::string &word = words[w];
+    if (style == case_style::pascal_case || style == case_style::camel_case) {
+      for (size_t i = 0; i < word.size(); i++) {
+        bool capital = (i == 0) && (style == case_style::pascal_case || w > 0);
+        result.push_back(capital ? to_upper(word[i]) : to_lower(word[i]));
+      }
+    } else {
+      bool upper = style == case_style::screaming_snake_case || style == case_style::screaming_kebab_case;
+      bool kebab = style == case_style::kebab_case || style == case_style::screaming_kebab_case;
+      if (w > 0) { result.push_back(kebab ? '-' : '_'); }
+      for (char c : word) { result.push_back(upper ? to_upper(c) : to_lower(c)); }
+    }
+  }
+  return result;
+}
+
+// The JSON key for a data member or an enumerator: an explicit rename wins,
+// then the rename_all of the enclosing structure or enumeration, then the C++
+// identifier.
+consteval std::string_view json_key_name(std::meta::info entity) {
+  std::meta::info rename_type = annotation_of_template(entity, ^^rename_t);
+  if (rename_type != std::meta::info{}) {
+    return std::define_static_string(std::string_view{
+        static_member_value<const char *>(rename_type, "key_data"),
+        static_member_value<size_t>(rename_type, "key_size")});
+  }
+  std::meta::info rename_all_type = annotation_of_template(std::meta::parent_of(entity), ^^rename_all_t);
+  if (rename_all_type != std::meta::info{}) {
+    case_style style = static_member_value<case_style>(rename_all_type, "style");
+    return std::define_static_string(apply_case_style(std::meta::identifier_of(entity), style));
+  }
+  return std::define_static_string(std::meta::identifier_of(entity));
+}
+
+// The keys accepted when deserializing a data member or an enumerator: the JSON
+// key first, followed by the aliases (if any), in declaration order.
+consteval std::vector<std::string_view> json_key_names(std::meta::info entity) {
+  std::vector<std::string_view> names{json_key_name(entity)};
+  for (std::meta::info ann : std::meta::annotations_of(entity)) {
+    std::meta::info type = std::meta::type_of(ann);
+    if (std::meta::has_template_arguments(type) && std::meta::template_of(type) == ^^alias_t) {
+      const std::string_view *keys = static_member_value<const std::string_view *>(type, "keys_data");
+      size_t count = static_member_value<size_t>(type, "keys_count");
+      for (size_t i = 0; i < count; i++) { names.push_back(keys[i]); }
+    }
+  }
+  return names;
+}
+
+// The structure type of a member annotated with flatten.
+consteval std::meta::info flattened_type(std::meta::info mem) {
+  if (std::meta::is_reference_type(std::meta::type_of(mem))) {
+    // A reference member could refer back to the enclosing structure: the
+    // flattening would never terminate.
+    throw std::meta::exception(u8"simdjson::flatten requires a member that is not a reference", mem);
+  }
+  std::meta::info type = std::meta::remove_cvref(std::meta::type_of(mem));
+  if (!std::meta::is_class_type(type)) {
+    throw std::meta::exception(u8"simdjson::flatten requires a member of class type", mem);
+  }
+  return type;
+}
+
+// The single data member of a structure annotated with transparent: the only
+// member that is not annotated with skip.
+consteval std::meta::info transparent_member(std::meta::info type) {
+  std::vector<std::meta::info> members;
+  for (std::meta::info mem : std::meta::nonstatic_data_members_of(type, std::meta::access_context::unchecked())) {
+    if (!has_annotation(mem, ^^skip_tag)) { members.push_back(mem); }
+  }
+  if (members.size() != 1) {
+    throw std::meta::exception(u8"simdjson::transparent requires exactly one data member (not counting skipped members)", type);
+  }
+  return members[0];
+}
+
+} // namespace detail
+
+// Returns the JSON key for a reflected data member (or enumerator).
+template <auto dm>
+consteval const char* get_json_key_name() {
+  return detail::json_key_name(dm).data();
+}
+
+} // namespace simdjson
+
+#endif // SIMDJSON_STATIC_REFLECTION
+#endif // SIMDJSON_ANNOTATIONS_H
+/* end file simdjson/annotations.h */

 #endif // SIMDJSON_GENERIC_BUILDER_DEPENDENCIES_H
 /* end file simdjson/generic/builder/dependencies.h */
@@ -38297,7 +45635,14 @@ simdjson_inline int leading_zeroes(uint64_t input_num) {

 /* result might be undefined when input_num is zero */
 simdjson_inline int count_ones(uint64_t input_num) {
+#if SIMDJSON_REGULAR_VISUAL_STUDIO
    return vaddv_u8(vcnt_u8(vcreate_u8(input_num)));
+#else
+   // if the system supports SVE or CSSC, __builtin_popcountll
+   // might be compiled to fewer single instructions. For CSSC,
+   // __builtin_popcountll is compiled to a single instruction.
+   return __builtin_popcountll(input_num);
+#endif// SIMDJSON_REGULAR_VISUAL_STUDIO
 }


@@ -38334,15 +45679,6 @@ simdjson_inline uint64_t zero_leading_bit(uint64_t rev_bits, int leading_zeroes)

 #endif

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2, uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
-  *result = value1 + value2;
-  return *result < value1;
-#else
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-#endif
-}

 } // unnamed namespace
 } // namespace arm64
@@ -38606,6 +45942,7 @@ namespace {
       return vget_lane_u64(
           vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(*this), 4)), 0);
     }
+    // Integer umaxv, not a float compare: flush-to-zero would treat a denormal as zero.
     simdjson_inline bool any() const { return vmaxvq_u32(vreinterpretq_u32_u8(*this)) != 0; }
   };

@@ -39240,7 +46577,7 @@ public:
 #if SIMDJSON_SUPPORTS_RANGES && SIMDJSON_SUPPORTS_CONCEPTS
   // Support for range-based appending (std::ranges::view, etc.)
   template <std::ranges::range R>
-requires (!std::is_convertible<R, std::string_view>::value && !require_custom_serialization<R>)
+requires (!std::is_convertible<R, std::string_view>::value && !concepts::optional_type<R> && !require_custom_serialization<R>)
   simdjson_inline void append(const R &range) noexcept;
 #endif
   /**
@@ -39254,6 +46591,15 @@ requires (!std::is_convertible<R, std::string_view>::value && !require_custom_se
    * There is no UTF-8 validation.
    */
   simdjson_inline void append_raw(const char *str, size_t len) noexcept;
+
+  /**
+   * Append exactly N characters from str. The length is a template parameter
+   * so the compiler can fully inline the memcpy with a compile-time-constant
+   * size, avoiding the libc call. Used for compile-time-constant keys in the
+   * reflection struct atom.
+   */
+  template <size_t N>
+  simdjson_inline void append_raw_n(const char *str) noexcept;
 #if SIMDJSON_EXCEPTIONS
   /**
    * Creates an std::string from the written JSON buffer.
@@ -39305,6 +46651,26 @@ requires (!std::is_convertible<R, std::string_view>::value && !require_custom_se
    */
   simdjson_inline size_t size() const noexcept;

+  // ============================================================
+  // Internal hooks for the position-as-local writer in json_builder.h.
+  // These exist so the reflection atom code can hold buffer pointer,
+  // position and capacity in registers across long write chains rather
+  // than reloading them after every char* write (strict aliasing
+  // forces those reloads when accessed via members of *this). User
+  // code should NOT call these directly.
+  // ============================================================
+  simdjson_inline char *unsafe_data() noexcept { return buffer.get(); }
+  simdjson_inline size_t unsafe_position() const noexcept { return position; }
+  simdjson_inline size_t unsafe_capacity() const noexcept { return capacity; }
+  simdjson_inline void unsafe_set_position(size_t p) noexcept { position = p; }
+  /// Make capacity available for at least `n` more bytes after the current
+  /// position. Returns false if the allocation failed.
+  simdjson_inline bool unsafe_grow(size_t needed_total_capacity) noexcept {
+    grow_buffer(needed_total_capacity);
+    return is_valid;
+  }
+  simdjson_inline bool unsafe_is_valid() const noexcept { return is_valid; }
+
 private:
   /**
    * Returns true if we can write at least upcoming_bytes bytes.
@@ -39318,7 +46684,7 @@ private:
    * If the allocation fails, is_valid is set to false. We expect
    * that this function would not be repeatedly called.
    */
-  simdjson_inline void grow_buffer(size_t desired_capacity);
+  inline void grow_buffer(size_t desired_capacity);

   /**
    * We use this helper function to make sure that is_valid is kept consistent.
@@ -39375,6 +46741,7 @@ simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initi
 /* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_STRING_BUILDER_H */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/builder/json_string_builder.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/concepts.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/annotations.h" */
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
 #if SIMDJSON_STATIC_REFLECTION

@@ -39392,64 +46759,370 @@ namespace simdjson {
 namespace arm64 {
 namespace builder {

-template <class T>
+// Forward-declare helpers defined in json_string_builder-inl.h so the
+// writer-based atom code below can call them (the -inl.h is not yet
+// included at the point this header is parsed; without these forwards,
+// name lookup falls back to the wrong outer namespace).
+namespace internal {
+simdjson_really_inline char *write_uint_jeaiii(char *p, uint64_t v) noexcept;
+simdjson_inline char *write_double(char *p, double v) noexcept;
+} // namespace internal
+simdjson_really_inline size_t write_string_escaped(const std::string_view input, char *out);
+
+SIMDJSON_PUSH_DISABLE_WARNINGS
+SIMDJSON_DISABLE_GCC_WARNING(-Warray-bounds)
+#if !defined(__clang__)
+SIMDJSON_DISABLE_GCC_WARNING(-Wstringop-overflow)
+#endif
+
+// =============================================================
+// `writer`: position-as-local hot-path writer used by the reflection
+// atom code below. Holds the buffer pointer, write position and
+// capacity in three fields that, once `writer` itself is a stack-local
+// in the caller and all atom() functions are inlined, become true
+// register-resident locals after SROA. Glaze achieves the same effect
+// by passing `B&& b, auto&& ix` through every helper. Holding `pos`
+// in a register (rather than as a member of string_builder) is what
+// breaks the strict-aliasing penalty on every char* write through the
+// buffer, which forces a reload of `b.position` and `b.capacity`
+// after every byte.
+//
+// basic_writer<false> (below) is the unchecked variant: the caller has
+// already reserved enough capacity for everything the write chain can
+// produce (see bound_detail::size_bound), so ensure() compiles away.
+// =============================================================
+template <bool Checked>
+struct basic_writer {
+  static constexpr bool checked = Checked;
+  char *ptr;        // buffer pointer (refreshed after a grow)
+  size_t pos;       // write position (local)
+  size_t cap;       // capacity (refreshed after a grow)
+  string_builder &sb;  // back-ref for grow / sync
+
+  // Snapshot string_builder state into a writer for the duration of
+  // a write chain.
+  simdjson_really_inline basic_writer(string_builder &builder) noexcept
+      : ptr(builder.unsafe_data())
+      , pos(builder.unsafe_position())
+      , cap(builder.unsafe_capacity())
+      , sb(builder) {}
+
+  // Write the local position back to the underlying string_builder.
+  // Caller is responsible for invoking before the writer is dropped
+  // (otherwise data is lost). Idempotent.
+  simdjson_really_inline void sync() noexcept {
+    sb.unsafe_set_position(pos);
+  }
+
+  // Ensure at least `n` more bytes of free capacity. Grows the
+  // underlying buffer if needed (rare path). Returns false on
+  // allocation failure.
+  simdjson_really_inline bool ensure(size_t n) noexcept {
+    // pos <= cap, and cap is the size of a live allocation, so pos + n
+    // cannot wrap when n is a small constant or a compile-time length.
+    // Callers passing a size derived from input (the string atoms) must
+    // bound it against pos themselves. Keep the `pos + n <= cap` form:
+    // `n <= cap - pos` is measurably slower once the serializer is inlined.
+    if (simdjson_likely(pos + n <= cap)) { return true; }
+    return grow_slow(n);
+  }
+
+  simdjson_never_inline bool grow_slow(size_t n) noexcept {
+    // Detect overflow.
+    // This is pedantic except maybe on 32-bit targets.
+    if (simdjson_unlikely(pos + n < pos)) return false;
+    sb.unsafe_set_position(pos);
+    // even if 2*capacity overflows, the (std::max) below will pick the needed value,
+    // so we do not need a separate overflow check here.
+    if (!sb.unsafe_grow((std::max)(cap * 2, pos + n))) {
+      // The string_builder freed its buffer and is now invalid (null buffer,
+      // zero capacity and position). Mirror that state so that every later
+      // ensure() fails too: callers only return from the current atom, and
+      // their callers keep writing.
+      ptr = nullptr;
+      pos = 0;
+      cap = 0;
+      return false;
+    }
+    ptr = sb.unsafe_data();
+    cap = sb.unsafe_capacity();
+    return true;
+  }
+};
+
+// The unchecked writer writes into a raw buffer that the caller sized with
+// serialized_size_bound: it never grows and needs no string_builder.
+template <>
+struct basic_writer<false> {
+  static constexpr bool checked = false;
+  char *ptr;
+  size_t pos;
+
+  simdjson_really_inline basic_writer(char *buffer, size_t position) noexcept
+      : ptr(buffer), pos(position) {}
+
+  simdjson_really_inline bool ensure(size_t) const noexcept { return true; }
+};
+
+using writer = basic_writer<true>;
+using unchecked_writer = basic_writer<false>;
+
+// Bytes reserved past the size bound for an unchecked writer: it may then
+// write a little past the end of what it produces (e.g., copy keys as whole
+// 16-byte blocks).
+inline constexpr size_t unchecked_slack = 64;
+
+consteval size_t padded_key_length(size_t length) {
+  return (length + 15) / 16 * 16;
+}
+
+// === Helper: invoke a string_builder member that writes variable-length
+// content (escape_and_append_with_quotes etc), syncing the writer's local
+// state before the call and reloading after. Used for string fields where
+// rewriting the entire SIMD escape path through the writer would be a much
+// bigger refactor. f may be user code (a with<Adapter> serializer) that
+// throws: the exception then propagates to the caller.
+template <class W, class F>
+simdjson_really_inline void call_through_string_builder(W &w, F &&f) noexcept(noexcept(f(w.sb))) {
+  w.sync();
+  f(w.sb);
+  w.ptr = w.sb.unsafe_data();
+  w.pos = w.sb.unsafe_position();
+  w.cap = w.sb.unsafe_capacity();
+}
+
+// Helpers implementing the serialization side of the annotations (see
+// simdjson/annotations.h) for reflected structures.
+namespace annotation_detail {
+
+// A member is serialized unless it is annotated with skip or skip_serializing.
+consteval bool is_serialized_member(std::meta::info dm) {
+  return !simdjson::detail::has_annotation(dm, ^^simdjson::detail::skip_tag)
+      && !simdjson::detail::has_annotation(dm, ^^simdjson::detail::skip_serializing_tag);
+}
+
+// False when the member has a skip_serializing_if<pred> annotation and
+// pred(value) is true.
+template <auto dm, typename V>
+simdjson_really_inline bool should_serialize(const V &value) {
+  constexpr std::meta::info skip_if_type = simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::skip_serializing_if_t);
+  if constexpr (skip_if_type != std::meta::info{}) {
+    using skip_if = typename [: skip_if_type :];
+    return !skip_if::predicate(value);
+  } else {
+    (void)value;
+    return true;
+  }
+}
+
+// Serialize a member value, through its with<Adapter> annotation when the
+// adapter provides a serialize function.
+template <auto dm, class W, typename V>
+simdjson_really_inline void atom_member(W &w, const V &value) {
+  constexpr std::meta::info with_type = simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::with_t);
+  if constexpr (with_type != std::meta::info{}) {
+    using adapter = typename [: with_type :]::adapter;
+    if constexpr (requires(string_builder &b) { adapter::serialize(b, value); }) {
+      call_through_string_builder(w, [&](string_builder &b) { adapter::serialize(b, value); });
+    } else {
+      atom(w, value);
+    }
+  } else {
+    atom(w, value);
+  }
+}
+
+// Write the "key":value pairs of the members of t (without the braces), each
+// preceded by a comma unless it is the first one. The members of a member
+// annotated with flatten are written in its place.
+template <class W, class T>
+simdjson_really_inline void atom_fields(W &w, const T &t, bool &first) {
+  // Per-field block: ensure key+value worst case, then write key + value
+  // through the writer's local pos. For arithmetic fields, the integer
+  // write happens directly via write_uint_jeaiii on w.ptr+w.pos, so pos
+  // never round-trips through memory.
+  template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+    if constexpr (is_serialized_member(dm)) {
+      if (should_serialize<dm>(t.[:dm:])) {
+        if constexpr (simdjson::detail::has_annotation(dm, ^^simdjson::detail::flatten_tag)) {
+          static_assert(std::meta::is_class_type(simdjson::detail::flattened_type(dm)));
+          using flattened = std::remove_cvref_t<decltype(t.[:dm:])>;
+          static_assert(!concepts::container_but_not_string<flattened> && !concepts::string_view_keyed_map<flattened> &&
+                        !concepts::appendable_containers<flattened> && !concepts::optional_type<flattened> &&
+                        !concepts::smart_pointer<flattened> && !std::is_same_v<flattened, std::string> &&
+                        !std::is_same_v<flattened, std::string_view> && !require_custom_serialization<flattened>,
+                        "simdjson::flatten requires a member whose type is a structure serialized member by member");
+          atom_fields(w, t.[:dm:], first);
+        } else {
+          // Copy the key as whole 16-byte blocks from a zero-padded copy (one
+          // load and one store); ensure() reserves the padded length, and the
+          // unchecked writer has slack past its bound. Prior related work:
+          // jsonifier copies a power-of-two padded key and advances the cursor
+          // by the real length (serialize_impl.hpp, packed_blitter,
+          // https://github.com/nihilai-collective/Jsonifier).
+          constexpr const char* key_name = simdjson::get_json_key_name<dm>();
+          constexpr size_t first_key_len = constevalutil::consteval_to_quoted_escaped(key_name).size() + 1;
+          constexpr size_t rest_key_len = first_key_len + 1;
+          constexpr auto first_key = std::define_static_string(
+              constevalutil::consteval_to_quoted_escaped(key_name) + ":" +
+              std::string(padded_key_length(first_key_len) - first_key_len, '\0'));
+          constexpr auto rest_key = std::define_static_string(
+              std::string(",") + constevalutil::consteval_to_quoted_escaped(key_name) + ":" +
+              std::string(padded_key_length(rest_key_len) - rest_key_len, '\0'));
+          if (!w.ensure(padded_key_length(rest_key_len))) { return; }
+          if (first) {
+            std::memcpy(w.ptr + w.pos, first_key, padded_key_length(first_key_len));
+            w.pos += first_key_len;
+          } else {
+            std::memcpy(w.ptr + w.pos, rest_key, padded_key_length(rest_key_len));
+            w.pos += rest_key_len;
+          }
+          first = false;
+          atom_member<dm>(w, t.[:dm:]);
+        }
+      }
+    }
+  };
+}
+
+} // namespace annotation_detail
+
+template <class W, class T>
   requires(concepts::container_but_not_string<T> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &t) {
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
   auto it = t.begin();
   auto end = t.end();
   if (it == end) {
-    b.append_raw("[]");
+    if (!w.ensure(2)) return;
+    std::memcpy(w.ptr + w.pos, "[]", 2);
+    w.pos += 2;
     return;
   }
-  b.append('[');
-  atom(b, *it);
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = '[';
+  atom(w, *it);
   ++it;
   for (; it != end; ++it) {
-    b.append(',');
-    atom(b, *it);
+    if (!w.ensure(1)) return;
+    w.ptr[w.pos++] = ',';
+    atom(w, *it);
   }
-  b.append(']');
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = ']';
 }

-template <class T>
+template <class W, class T>
   requires(std::is_same_v<T, std::string> ||
            std::is_same_v<T, std::string_view> ||
            std::is_same_v<T, const char *> ||
            std::is_same_v<T, char>)
-constexpr void atom(string_builder &b, const T &t) {
-  b.escape_and_append_with_quotes(t);
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
+  // Inline the escape path through the writer so we never round-trip
+  // pos through memory for string fields (Twitter is dominated by
+  // these -- sync/reload around each string was a real cost).
+  std::string_view input;
+  if constexpr (std::is_same_v<T, char>) {
+    input = std::string_view(&t, 1);
+  } else {
+    input = std::string_view(t);
+  }
+  // Worst-case escape: every byte expands to \uXXXX (6 chars), plus 2 quotes.
+  // Guard against w.pos + 2 + 6 * input.size() wrapping for huge inputs -- if
+  // it wrapped to a small value, ensure() would spuriously succeed and the
+  // subsequent escape would overflow the buffer. max - w.pos cannot wrap, and
+  // size < (max - pos) / 6 implies pos + 6 * size + 6 <= max.
+  // Note that this is pedantic except maybe on 32-bit targets.
+  if constexpr (W::checked) {
+    if (simdjson_unlikely(input.size() >= ((std::numeric_limits<size_t>::max)() - w.pos) / 6)) { return; }
+    if (!w.ensure(2 + 6 * input.size())) { return; }
+  }
+  w.ptr[w.pos++] = '"';
+  w.pos += write_string_escaped(input, w.ptr + w.pos);
+  w.ptr[w.pos++] = '"';
 }

-template <concepts::string_view_keyed_map T>
+template <class W, concepts::string_view_keyed_map T>
   requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &m) {
+simdjson_really_inline constexpr void atom(W &w, const T &m) {
   if (m.empty()) {
-    b.append_raw("{}");
+    if (!w.ensure(2)) return;
+    std::memcpy(w.ptr + w.pos, "{}", 2);
+    w.pos += 2;
     return;
   }
-  b.append('{');
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = '{';
   bool first = true;
   for (const auto& [key, value] : m) {
     if (!first) {
-      b.append(',');
+      if (!w.ensure(1)) return;
+      w.ptr[w.pos++] = ',';
     }
     first = false;
-    // Keys must be convertible to string_view per the concept
-    b.escape_and_append_with_quotes(key);
-    b.append(':');
-    atom(b, value);
+    // Keys must be convertible to string_view per the concept.
+    std::string_view key_sv(key);
+    // Guard against w.pos + 3 + 6 * key_sv.size() wrapping for huge keys, if
+    // it wrapped to a small value, ensure() would spuriously succeed and the
+    // subsequent escape would overflow the buffer. max - w.pos cannot wrap, and
+    // size < (max - pos) / 6 implies pos + 6 * size + 6 <= max.
+    // Note that this is pedantic except maybe on 32-bit targets.
+    if constexpr (W::checked) {
+      if (simdjson_unlikely(key_sv.size() >= ((std::numeric_limits<size_t>::max)() - w.pos) / 6)) { return; }
+      if (!w.ensure(2 + 6 * key_sv.size() + 1)) { return; }
+    }
+    w.ptr[w.pos++] = '"';
+    w.pos += write_string_escaped(key_sv, w.ptr + w.pos);
+    w.ptr[w.pos++] = '"';
+    w.ptr[w.pos++] = ':';
+    atom(w, value);
   }
-  b.append('}');
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = '}';
 }


-template<typename number_type,
+template<class W, typename number_type,
          typename = typename std::enable_if<std::is_arithmetic<number_type>::value && !std::is_same_v<number_type, char>>::type>
-constexpr void atom(string_builder &b, const number_type t) {
-  b.append(t);
+simdjson_really_inline constexpr void atom(W &w, const number_type t) {
+  // Booleans / floats: defer to string_builder (rare path; keeps writer hot
+  // path free of float-formatter machinery). For integers, write directly
+  // via jeaiii using local pos.
+  if constexpr (std::is_same_v<number_type, bool>) {
+    if (t) {
+      if (!w.ensure(4)) return;
+      std::memcpy(w.ptr + w.pos, "true", 4);
+      w.pos += 4;
+    } else {
+      if (!w.ensure(5)) return;
+      std::memcpy(w.ptr + w.pos, "false", 5);
+      w.pos += 5;
+    }
+  } else if constexpr (std::is_floating_point_v<number_type>) {
+    if constexpr (W::checked) {
+      call_through_string_builder(w, [&](string_builder &b) { b.append(t); });
+    } else {
+      w.pos = size_t(internal::write_double(w.ptr + w.pos, double(t)) - w.ptr);
+    }
+  } else if constexpr (std::is_unsigned_v<number_type>) {
+    if (!w.ensure(20)) return;
+    char *end = internal::write_uint_jeaiii(
+        w.ptr + w.pos, static_cast<uint64_t>(t));
+    w.pos = static_cast<size_t>(end - w.ptr);
+  } else {
+    // signed integral
+    if (!w.ensure(20)) return;
+    using U = typename std::make_unsigned<number_type>::type;
+    bool negative = t < 0;
+    U pv = negative ? U(0) - static_cast<U>(t) : static_cast<U>(t);
+    w.ptr[w.pos] = '-';
+    w.pos += negative;
+    char *end = internal::write_uint_jeaiii(
+        w.ptr + w.pos, static_cast<uint64_t>(pv));
+    w.pos = static_cast<size_t>(end - w.ptr);
+  }
 }

-template <class T>
+template <class W, class T>
   requires(std::is_class_v<T> && !concepts::container_but_not_string<T> &&
            !concepts::string_view_keyed_map<T> &&
            !concepts::optional_type<T> &&
@@ -39459,92 +47132,259 @@ template <class T>
            !std::is_same_v<T, std::string_view> &&
            !std::is_same_v<T, const char*> &&
            !std::is_same_v<T, char> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &t) {
-  int i = 0;
-  b.append('{');
-  template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
-    if (i != 0)
-      b.append(',');
-    constexpr auto key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(dm)));
-    b.append_raw(key);
-    b.append(':');
-    atom(b, t.[:dm:]);
-    i++;
-  };
-  b.append('}');
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
+  if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag)) {
+    // A transparent structure is serialized as its single member.
+    constexpr auto dm = simdjson::detail::transparent_member(^^T);
+    annotation_detail::atom_member<dm>(w, t.[:dm:]);
+  } else {
+    bool first = true;
+    if (!w.ensure(1)) { return; }
+    w.ptr[w.pos++] = '{';
+    annotation_detail::atom_fields(w, t, first);
+    if (!w.ensure(1)) { return; }
+    w.ptr[w.pos++] = '}';
+  }
 }

 // Support for optional types (std::optional, etc.)
-template <concepts::optional_type T>
+template <class W, concepts::optional_type T>
   requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &opt) {
+simdjson_really_inline constexpr void atom(W &w, const T &opt) {
   if (opt) {
-    atom(b, opt.value());
+    atom(w, opt.value());
   } else {
-    b.append_raw("null");
+    if (!w.ensure(4)) return;
+    std::memcpy(w.ptr + w.pos, "null", 4);
+    w.pos += 4;
   }
 }

 // Support for smart pointers (std::unique_ptr, std::shared_ptr, etc.)
-template <concepts::smart_pointer T>
+template <class W, concepts::smart_pointer T>
   requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &ptr) {
+simdjson_really_inline constexpr void atom(W &w, const T &ptr) {
   if (ptr) {
-    atom(b, *ptr);
+    atom(w, *ptr);
   } else {
-    b.append_raw("null");
+    if (!w.ensure(4)) return;
+    std::memcpy(w.ptr + w.pos, "null", 4);
+    w.pos += 4;
   }
 }

 // Support for enums - serialize as string representation using expand approach from P2996R12
-template <typename T>
+template <class W, typename T>
   requires(std::is_enum_v<T> && !require_custom_serialization<T>)
-void atom(string_builder &b, const T &e) {
+simdjson_really_inline void atom(W &w, const T &e) {
 #if SIMDJSON_STATIC_REFLECTION
-  constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
+  static constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
   template for (constexpr auto enum_val : enumerators) {
-    constexpr auto enum_str = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(enum_val)));
+    constexpr auto enum_str = std::define_static_string(constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<enum_val>()));
+    constexpr size_t enum_str_len = std::char_traits<char>::length(enum_str);
     if (e == [:enum_val:]) {
-      b.append_raw(enum_str);
+      if (!w.ensure(enum_str_len)) return;
+      std::memcpy(w.ptr + w.pos, enum_str, enum_str_len);
+      w.pos += enum_str_len;
       return;
     }
   };
   // Fallback to integer if enum value not found
-  atom(b, static_cast<std::underlying_type_t<T>>(e));
+  atom(w, static_cast<std::underlying_type_t<T>>(e));
 #else
   // Fallback: serialize as integer if reflection not available
-  atom(b, static_cast<std::underlying_type_t<T>>(e));
+  atom(w, static_cast<std::underlying_type_t<T>>(e));
 #endif
 }

 // Support for appendable containers that don't have operator[] (sets, etc.)
-template <concepts::appendable_containers T>
+template <class W, concepts::appendable_containers T>
   requires(!concepts::container_but_not_string<T> && !concepts::string_view_keyed_map<T> &&
            !concepts::optional_type<T> && !concepts::smart_pointer<T> &&
            !std::is_same_v<T, std::string> &&
            !std::is_same_v<T, std::string_view> && !std::is_same_v<T, const char*> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &container) {
+simdjson_really_inline constexpr void atom(W &w, const T &container) {
   if (container.empty()) {
-    b.append_raw("[]");
+    if (!w.ensure(2)) return;
+    std::memcpy(w.ptr + w.pos, "[]", 2);
+    w.pos += 2;
     return;
   }
-  b.append('[');
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = '[';
   bool first = true;
   for (const auto& item : container) {
     if (!first) {
-      b.append(',');
+      if (!w.ensure(1)) return;
+      w.ptr[w.pos++] = ',';
     }
     first = false;
-    atom(b, item);
+    atom(w, item);
+  }
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = ']';
+}
+
+// =============================================================
+// Size bound: an upper bound on the number of bytes that atom(w, t) writes.
+// Computing it first lets append() reserve the capacity once and then run
+// the whole write chain through an unchecked_writer, without a capacity
+// check before every write. It mirrors the atom() overloads above.
+//
+// Prior related work: jsonifier sizes the document first, counting 6 bytes
+// per string byte, resizes once to that bound plus slack, and writes with
+// no capacity check on each store. to_json() does that resize through
+// std::string::resize_and_overwrite. See serializer.hpp and
+// serialize_impl.hpp in https://github.com/nihilai-collective/Jsonifier
+// and https://nihilai-collective.net/serialization.
+// =============================================================
+namespace bound_detail {
+
+// Whether size_bound covers everything that atom() writes for T: not when a
+// member is serialized by a with<Adapter> serializer, which writes an unknown
+// amount through the string_builder.
+template <class T>
+consteval bool is_bounded() {
+  if constexpr (require_custom_serialization<T>) {
+    return false;
+  } else if constexpr (std::is_same_v<T, std::string> || std::is_same_v<T, std::string_view> ||
+                       std::is_same_v<T, const char *> || std::is_arithmetic_v<T> || std::is_enum_v<T>) {
+    return true;
+  } else if constexpr (concepts::optional_type<T> || concepts::smart_pointer<T>) {
+    return is_bounded<std::remove_cvref_t<decltype(*std::declval<const T &>())>>();
+  } else if constexpr (concepts::string_view_keyed_map<T>) {
+    return is_bounded<std::remove_cvref_t<typename T::mapped_type>>();
+  } else if constexpr (concepts::container_but_not_string<T> || concepts::appendable_containers<T>) {
+    return is_bounded<std::remove_cvref_t<std::ranges::range_value_t<T>>>();
+  } else {
+    bool bounded = true;
+    template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+      if constexpr (annotation_detail::is_serialized_member(dm)) {
+        bounded = bounded && simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::with_t) == std::meta::info{} &&
+                  is_bounded<std::remove_cvref_t<decltype(std::declval<const T &>().[:dm:])>>();
+      }
+    };
+    return bounded;
+  }
+}
+
+template <class T>
+consteval size_t enum_bound() {
+  size_t bound = 20; // the integer fallback
+  template for (constexpr auto enum_val : std::define_static_array(std::meta::enumerators_of(^^T))) {
+    constexpr size_t len = std::char_traits<char>::length(std::define_static_string(
+        constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<enum_val>())));
+    bound = (std::max)(bound, len);
+  };
+  return bound;
+}
+
+template <class T>
+simdjson_really_inline size_t size_bound(const T &t) noexcept;
+
+// Bound for the "key":value pairs of a structure, commas included.
+template <class T>
+simdjson_really_inline size_t fields_bound(const T &t) noexcept {
+  size_t bound = 0;
+  template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+    if constexpr (annotation_detail::is_serialized_member(dm)) {
+      if constexpr (simdjson::detail::has_annotation(dm, ^^simdjson::detail::flatten_tag)) {
+        bound += fields_bound(t.[:dm:]);
+      } else {
+        constexpr size_t rest_key_len = constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<dm>()).size() + 2;
+        bound += rest_key_len + size_bound(t.[:dm:]);
+      }
+    }
+  };
+  return bound;
+}
+
+template <class T>
+simdjson_really_inline size_t size_bound([[maybe_unused]] const T &t) noexcept {
+  if constexpr (std::is_same_v<T, char>) {
+    return 2 + 6;
+  } else if constexpr (std::is_same_v<T, std::string> || std::is_same_v<T, std::string_view> ||
+                       std::is_same_v<T, const char *>) {
+    // Every byte may become \uXXXX, plus the quotes.
+    return 2 + 6 * std::string_view(t).size();
+  } else if constexpr (std::is_same_v<T, bool>) {
+    return 5;
+  } else if constexpr (std::is_floating_point_v<T>) {
+    return simdjson::internal::to_chars_buffer_size;
+  } else if constexpr (std::is_arithmetic_v<T>) {
+    return 20;
+  } else if constexpr (std::is_enum_v<T>) {
+    return enum_bound<T>();
+  } else if constexpr (concepts::optional_type<T> || concepts::smart_pointer<T>) {
+    return t ? size_bound(*t) : 4;
+  } else if constexpr (concepts::string_view_keyed_map<T>) {
+    size_t bound = 2;
+    for (const auto &[key, value] : t) {
+      // comma, quotes, colon
+      bound += 4 + 6 * std::string_view(key).size() + size_bound(value);
+    }
+    return bound;
+  } else if constexpr (concepts::container_but_not_string<T> || concepts::appendable_containers<T>) {
+    using value_type = std::remove_cvref_t<std::ranges::range_value_t<T>>;
+    if constexpr (std::is_arithmetic_v<value_type> && !std::is_same_v<value_type, char>) {
+      // A fixed bound per element: no need to visit them.
+      return 2 + size_t(std::ranges::distance(t)) * (1 + size_bound(value_type{}));
+    } else {
+      size_t bound = 2;
+#if defined(__GNUC__) && !defined(__clang__)
+#pragma GCC novector // a vector loop is slower on short containers
+#endif
+#pragma GCC unroll 4
+      for (const auto &item : t) {
+        bound += 1 + size_bound(item);
+      }
+      return bound;
+    }
+  } else if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag)) {
+    constexpr auto dm = simdjson::detail::transparent_member(^^T);
+    return size_bound(t.[:dm:]);
+  } else {
+    return 2 + fields_bound(t);
+  }
+}
+
+} // namespace bound_detail
+
+// Write t through an unchecked writer when its size bound is available,
+// reserving that many bytes first, and through the checked writer otherwise.
+template <class T>
+simdjson_really_inline void append_bounded(string_builder &b, const T &t) {
+  // On 32-bit systems, the bound could overflow: keep the checked writer.
+  if constexpr (sizeof(size_t) >= 8 && bound_detail::is_bounded<T>()) {
+    const size_t bound = bound_detail::size_bound(t) + unchecked_slack;
+    const size_t pos = b.unsafe_position();
+    // The bound is a sum of in-memory sizes times a small constant: it cannot
+    // overflow on a 64-bit system. Be pedantic elsewhere.
+    if (sizeof(size_t) >= 8 || bound <= (std::numeric_limits<size_t>::max)() - pos) {
+      const size_t cap = b.unsafe_capacity();
+      // Grow geometrically so that many small appends stay amortized.
+      if (pos + bound <= cap || b.unsafe_grow((std::max)(cap * 2, pos + bound))) {
+        unchecked_writer w(b.unsafe_data(), pos);
+        atom(w, t);
+        b.unsafe_set_position(w.pos);
+      }
+      return;
+    }
   }
-  b.append(']');
+  writer w(b);
+  atom(w, t);
+  w.sync();
 }

-// append functions that delegate to atom functions for primitive types
+// append() -- top-level entry. Each overload constructs a stack-local
+// writer, runs atom(w, t) through the inlined call chain, then syncs
+// the local position back into the string_builder.
 template <class T>
   requires(std::is_arithmetic_v<T> && !std::is_same_v<T, char>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  writer w(b);
+  atom(w, t);
+  w.sync();
 }

 template <class T>
@@ -39552,20 +47392,22 @@ template <class T>
            std::is_same_v<T, std::string_view> ||
            std::is_same_v<T, const char *> ||
            std::is_same_v<T, char>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  writer w(b);
+  atom(w, t);
+  w.sync();
 }

 template <concepts::optional_type T>
   requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  append_bounded(b, t);
 }

 template <concepts::smart_pointer T>
   requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  append_bounded(b, t);
 }

 template <concepts::appendable_containers T>
@@ -39573,14 +47415,14 @@ template <concepts::appendable_containers T>
            !concepts::optional_type<T> && !concepts::smart_pointer<T> &&
            !std::is_same_v<T, std::string> &&
            !std::is_same_v<T, std::string_view> && !std::is_same_v<T, const char*> && !require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  append_bounded(b, t);
 }

 template <concepts::string_view_keyed_map T>
   requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  append_bounded(b, t);
 }

 // works for struct
@@ -39594,39 +47436,15 @@ template <class Z>
            !std::is_same_v<Z, std::string_view> &&
            !std::is_same_v<Z, const char*> &&
            !std::is_same_v<Z, char> && !require_custom_serialization<Z>)
-void append(string_builder &b, const Z &z) {
-  int i = 0;
-  b.append('{');
-  template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^Z, std::meta::access_context::unchecked()))) {
-    if (i != 0)
-      b.append(',');
-    constexpr auto key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(dm)));
-    b.append_raw(key);
-    b.append(':');
-    atom(b, z.[:dm:]);
-    i++;
-  };
-  b.append('}');
+simdjson_inline void append(string_builder &b, const Z &z) {
+  append_bounded(b, z);
 }

 // works for container that have begin() and end() iterators
 template <class Z>
   requires(concepts::container_but_not_string<Z> && !require_custom_serialization<Z>)
-void append(string_builder &b, const Z &z) {
-  auto it = z.begin();
-  auto end = z.end();
-  if (it == end) {
-    b.append_raw("[]");
-    return;
-  }
-  b.append('[');
-  atom(b, *it);
-  ++it;
-  for (; it != end; ++it) {
-    b.append(',');
-    atom(b, *it);
-  }
-  b.append(']');
+simdjson_inline void append(string_builder &b, const Z &z) {
+  append_bounded(b, z);
 }

 template <class Z>
@@ -39637,22 +47455,40 @@ void append(string_builder &b, const Z &z) {


 template <class Z>
-simdjson_warn_unused simdjson_result<std::string> to_json_string(const Z &z, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
-  string_builder b(initial_capacity);
-  append(b, z);
-  std::string_view s;
-  if(auto e = b.view().get(s); e) { return e; }
-  return std::string(s);
+simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
+  if constexpr (sizeof(size_t) >= 8 && bound_detail::is_bounded<Z>()) {
+    // Write straight into s, sized by the bound: no intermediate buffer, no copy.
+    // Prior related work: jsonifier's serializeJson resizes once through
+    // resize_and_overwrite (serializer.hpp).
+    (void)initial_capacity;
+    const size_t bound = bound_detail::size_bound(z) + unchecked_slack;
+    auto write = [&z](char *p) noexcept {
+      unchecked_writer w(p, 0);
+      atom(w, z);
+      return w.pos;
+    };
+#if defined(__cpp_lib_string_resize_and_overwrite) && __cpp_lib_string_resize_and_overwrite >= 202110L
+    s.resize_and_overwrite(bound, [&write](char *p, size_t) noexcept { return write(p); });
+#else
+    s.resize(bound);
+    s.resize(write(s.data()));
+#endif
+    return SUCCESS;
+  } else {
+    string_builder b(initial_capacity);
+    append(b, z);
+    std::string_view view;
+    if(auto e = b.view().get(view); e) { return e; }
+    s.assign(view);
+    return SUCCESS;
+  }
 }

 template <class Z>
-simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
-  string_builder b(initial_capacity);
-  append(b, z);
-  std::string_view view;
-  if(auto e = b.view().get(view); e) { return e; }
-  s.assign(view);
-  return SUCCESS;
+simdjson_warn_unused simdjson_result<std::string> to_json_string(const Z &z, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
+  std::string s;
+  if(auto e = to_json(z, s, initial_capacity); e) { return e; }
+  return s;
 }

 template <class Z>
@@ -39665,40 +47501,41 @@ string_builder& operator<<(string_builder& b, const Z& z) {
 template<constevalutil::fixed_string... FieldNames, typename T>
   requires(std::is_class_v<T> && (sizeof...(FieldNames) > 0))
 void extract_from(string_builder &b, const T &obj) {
-  // Helper to check if a field name matches any of the requested fields
-  auto should_extract = [](std::string_view field_name) constexpr -> bool {
-    return ((FieldNames.view() == field_name) || ...);
-  };
-
-  b.append('{');
+  writer w(b);
+  if (!w.ensure(1)) { w.sync(); return; }
+  w.ptr[w.pos++] = '{';
   bool first = true;
-
   // Iterate through all members of T using reflection
-  template for (constexpr auto mem : std::define_static_array(
-      std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
-
+  static constexpr auto members = std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()));
+  template for (constexpr auto mem : members) {
     if constexpr (std::meta::is_public(mem)) {
-      constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
+      static constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));

       // Only serialize this field if it's in our list of requested fields
-      if constexpr (should_extract(key)) {
-        if (!first) {
-          b.append(',');
+      if constexpr (((FieldNames.view() == key) || ...)) {
+        static constexpr auto first_key = std::define_static_string(
+            constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)) + ":");
+        static constexpr auto rest_key = std::define_static_string(
+            std::string(",") + constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)) + ":");
+        constexpr size_t first_key_len = std::char_traits<char>::length(first_key);
+        constexpr size_t rest_key_len = std::char_traits<char>::length(rest_key);
+        if (!w.ensure(rest_key_len)) { w.sync(); return; }
+        if (first) {
+          std::memcpy(w.ptr + w.pos, first_key, first_key_len);
+          w.pos += first_key_len;
+        } else {
+          std::memcpy(w.ptr + w.pos, rest_key, rest_key_len);
+          w.pos += rest_key_len;
         }
         first = false;
-
-        // Serialize the key
-        constexpr auto quoted_key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)));
-        b.append_raw(quoted_key);
-        b.append(':');
-
-        // Serialize the value
-        atom(b, obj.[:mem:]);
+        atom(w, obj.[:mem:]);
       }
     }
   };

-  b.append('}');
+  if (!w.ensure(1)) { w.sync(); return; }
+  w.ptr[w.pos++] = '}';
+  w.sync();
 }

 template<constevalutil::fixed_string... FieldNames, typename T>
@@ -39711,25 +47548,18 @@ simdjson_warn_unused simdjson_result<std::string> extract_from(const T &obj, siz
   return std::string(s);
 }

+SIMDJSON_POP_DISABLE_WARNINGS
+
 } // namespace builder
 } // namespace arm64
 // Alias the function template to 'to' in the global namespace
 template <class Z>
 simdjson_warn_unused simdjson_result<std::string> to_json(const Z &z, size_t initial_capacity = arm64::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
-  arm64::builder::string_builder b(initial_capacity);
-  arm64::builder::append(b, z);
-  std::string_view s;
-  if(auto e = b.view().get(s); e) { return e; }
-  return std::string(s);
+  return arm64::builder::to_json_string(z, initial_capacity);
 }
 template <class Z>
 simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = arm64::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
-  arm64::builder::string_builder b(initial_capacity);
-  arm64::builder::append(b, z);
-  std::string_view view;
-  if(auto e = b.view().get(view); e) { return e; }
-  s.assign(view);
-  return SUCCESS;
+  return arm64::builder::to_json(z, s, initial_capacity);
 }
 // Global namespace function for extract_from
 template<constevalutil::fixed_string... FieldNames, typename T>
@@ -39875,6 +47705,7 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
 /* including simdjson/generic/builder/json_string_builder-inl.h for arm64: #include "simdjson/generic/builder/json_string_builder-inl.h" */
 /* begin file simdjson/generic/builder/json_string_builder-inl.h for arm64 */
 #include <array>
+#include <cmath>
 #include <cstring>
 #include <limits>
 #include <type_traits>
@@ -39909,6 +47740,11 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
 #define SIMDJSON_EXPERIMENTAL_HAS_LSX 1
 #endif
 #endif
+#if defined(__loongarch_asx)
+#ifndef SIMDJSON_EXPERIMENTAL_HAS_LASX
+#define SIMDJSON_EXPERIMENTAL_HAS_LASX 1
+#endif
+#endif
 #if defined(__riscv_v_intrinsic) && __riscv_v_intrinsic >= 11000 &&            \
     defined(__riscv_vector)
 #ifndef SIMDJSON_EXPERIMENTAL_HAS_RVV
@@ -39928,6 +47764,9 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
 #endif
 #if SIMDJSON_EXPERIMENTAL_HAS_SSE2
 #include <emmintrin.h>
+#if defined(__AVX2__)
+#include <immintrin.h>
+#endif
 #ifdef _MSC_VER
 #include <intrin.h>
 #endif
@@ -39935,6 +47774,9 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
 #if SIMDJSON_EXPERIMENTAL_HAS_LSX
 #include <lsxintrin.h>
 #endif
+#if SIMDJSON_EXPERIMENTAL_HAS_LASX
+#include <lasxintrin.h>
+#endif
 #if SIMDJSON_EXPERIMENTAL_HAS_RVV
 #include <riscv_vector.h>
 #endif
@@ -39984,105 +47826,6 @@ inline bool has_json_escapable_byte(uint64_t x) {

 **/

-SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline bool
-simple_needs_escaping(std::string_view v) {
-  for (char c : v) {
-    // a table lookup is faster than a series of comparisons
-    if (json_quotable_character[static_cast<uint8_t>(c)]) {
-      return true;
-    }
-  }
-  return false;
-}
-
-#if SIMDJSON_EXPERIMENTAL_HAS_NEON
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
-  if (view.size() < 16) {
-    return simple_needs_escaping(view);
-  }
-  size_t i = 0;
-  uint8x16_t running = vdupq_n_u8(0);
-  uint8x16_t v34 = vdupq_n_u8(34);
-  uint8x16_t v92 = vdupq_n_u8(92);
-
-  for (; i + 15 < view.size(); i += 16) {
-    uint8x16_t word = vld1q_u8((const uint8_t *)view.data() + i);
-    running = vorrq_u8(running, vceqq_u8(word, v34));
-    running = vorrq_u8(running, vceqq_u8(word, v92));
-    running = vorrq_u8(running, vcltq_u8(word, vdupq_n_u8(32)));
-  }
-  if (i < view.size()) {
-    uint8x16_t word =
-        vld1q_u8((const uint8_t *)view.data() + view.length() - 16);
-    running = vorrq_u8(running, vceqq_u8(word, v34));
-    running = vorrq_u8(running, vceqq_u8(word, v92));
-    running = vorrq_u8(running, vcltq_u8(word, vdupq_n_u8(32)));
-  }
-  return vmaxvq_u32(vreinterpretq_u32_u8(running)) != 0;
-}
-#elif SIMDJSON_EXPERIMENTAL_HAS_SSE2
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
-  if (view.size() < 16) {
-    return simple_needs_escaping(view);
-  }
-  size_t i = 0;
-  __m128i running = _mm_setzero_si128();
-  for (; i + 15 < view.size(); i += 16) {
-
-    __m128i word =
-        _mm_loadu_si128(reinterpret_cast<const __m128i *>(view.data() + i));
-    running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(34)));
-    running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(92)));
-    running = _mm_or_si128(
-        running, _mm_cmpeq_epi8(_mm_subs_epu8(word, _mm_set1_epi8(31)),
-                                _mm_setzero_si128()));
-  }
-  if (i < view.size()) {
-    __m128i word = _mm_loadu_si128(
-        reinterpret_cast<const __m128i *>(view.data() + view.length() - 16));
-    running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(34)));
-    running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(92)));
-    running = _mm_or_si128(
-        running, _mm_cmpeq_epi8(_mm_subs_epu8(word, _mm_set1_epi8(31)),
-                                _mm_setzero_si128()));
-  }
-  return _mm_movemask_epi8(running) != 0;
-}
-#elif SIMDJSON_EXPERIMENTAL_HAS_PPC64
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
-  if (view.size() < 16) {
-    return simple_needs_escaping(view);
-  }
-  size_t i = 0;
-  __vector unsigned char running = vec_splats((unsigned char)0);
-  __vector unsigned char v34 = vec_splats((unsigned char)34);
-  __vector unsigned char v92 = vec_splats((unsigned char)92);
-  __vector unsigned char v32 = vec_splats((unsigned char)32);
-
-  for (; i + 15 < view.size(); i += 16) {
-    __vector unsigned char word =
-        vec_vsx_ld(0, reinterpret_cast<const unsigned char *>(view.data() + i));
-    running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v34));
-    running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v92));
-    running = vec_or(running,
-        (__vector unsigned char)vec_cmplt(word, v32));
-  }
-  if (i < view.size()) {
-    __vector unsigned char word = vec_vsx_ld(
-        0, reinterpret_cast<const unsigned char *>(view.data() + view.length() - 16));
-    running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v34));
-    running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v92));
-    running = vec_or(running,
-        (__vector unsigned char)vec_cmplt(word, v32));
-  }
-  return !vec_all_eq(running, vec_splats((unsigned char)0));
-}
-#else
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
-  return simple_needs_escaping(view);
-}
-#endif
-
 // Scalar fallback for finding next quotable character
 SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline size_t
 find_next_json_quotable_character_scalar(const std::string_view view,
@@ -40173,6 +47916,51 @@ find_next_json_quotable_character(const std::string_view view,
   size_t current = len - remaining;
   return find_next_json_quotable_character_scalar(view, current);
 }
+#elif SIMDJSON_EXPERIMENTAL_HAS_LASX
+simdjson_inline size_t
+find_next_json_quotable_character(const std::string_view view,
+                                  size_t location) noexcept {
+  const size_t len = view.size();
+  const uint8_t *ptr =
+      reinterpret_cast<const uint8_t *>(view.data()) + location;
+  size_t remaining = len - location;
+
+  // SIMD constants for characters requiring escape
+  __m256i v34 = __lasx_xvreplgr2vr_b(34);  // '"'
+  __m256i v92 = __lasx_xvreplgr2vr_b(92);  // '\\'
+  __m256i v32 = __lasx_xvreplgr2vr_b(32);  // control char threshold
+
+  while (remaining >= 32) {
+    __m256i word = __lasx_xvld(ptr, 0);
+
+    // Check for quotable characters: '"', '\\', or control chars (< 32)
+    __m256i needs_escape = __lasx_xvseq_b(word, v34);
+    needs_escape = __lasx_xvor_v(needs_escape, __lasx_xvseq_b(word, v92));
+    needs_escape = __lasx_xvor_v(needs_escape, __lasx_xvslt_bu(word, v32));
+
+    if (!__lasx_xbz_v(needs_escape)) {
+      // Found a quotable character - locate it via the four 64-bit lanes
+      uint64_t lane0 = __lasx_xvpickve2gr_du(needs_escape, 0);
+      uint64_t lane1 = __lasx_xvpickve2gr_du(needs_escape, 1);
+      uint64_t lane2 = __lasx_xvpickve2gr_du(needs_escape, 2);
+      uint64_t lane3 = __lasx_xvpickve2gr_du(needs_escape, 3);
+      size_t offset = ptr - reinterpret_cast<const uint8_t *>(view.data());
+      if (lane0 != 0) {
+        return offset + trailing_zeroes(lane0) / 8;
+      } else if (lane1 != 0) {
+        return offset + 8 + trailing_zeroes(lane1) / 8;
+      } else if (lane2 != 0) {
+        return offset + 16 + trailing_zeroes(lane2) / 8;
+      } else {
+        return offset + 24 + trailing_zeroes(lane3) / 8;
+      }
+    }
+    ptr += 32;
+    remaining -= 32;
+  }
+  size_t current = len - remaining;
+  return find_next_json_quotable_character_scalar(view, current);
+}
 #elif SIMDJSON_EXPERIMENTAL_HAS_LSX
 simdjson_inline size_t
 find_next_json_quotable_character(const std::string_view view,
@@ -40328,6 +48116,254 @@ SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline void escape_json_char(char c, char *&o
   }
 }

+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2 || SIMDJSON_EXPERIMENTAL_HAS_NEON
+#define SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE 1
+#endif
+
+#if SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2
+
+using escape_vector = __m128i;
+// Mask bits that each input byte contributes to escape_bitmask().
+static constexpr unsigned escape_mask_bits = 1;
+
+simdjson_inline escape_vector escape_load16(const uint8_t *p) noexcept {
+  return _mm_loadu_si128(reinterpret_cast<const __m128i *>(p));
+}
+
+simdjson_inline void escape_store16(char *out, escape_vector v) noexcept {
+  _mm_storeu_si128(reinterpret_cast<__m128i *>(out), v);
+}
+
+// Builds a vector whose bytes 0..7 come from a and bytes 8..15 from b.
+simdjson_inline escape_vector escape_load8x2(const uint8_t *a,
+                                             const uint8_t *b) noexcept {
+  return _mm_unpacklo_epi64(
+      _mm_loadl_epi64(reinterpret_cast<const __m128i *>(a)),
+      _mm_loadl_epi64(reinterpret_cast<const __m128i *>(b)));
+}
+
+// Builds a vector whose bytes 0..3 come from a and bytes 4..7 from b. The
+// upper half is unspecified; callers only look at the low eight bits.
+simdjson_inline escape_vector escape_load4x2(const uint8_t *a,
+                                             const uint8_t *b) noexcept {
+  int32_t a32, b32;
+  memcpy(&a32, a, 4);
+  memcpy(&b32, b, 4);
+  return _mm_unpacklo_epi32(_mm_cvtsi32_si128(a32), _mm_cvtsi32_si128(b32));
+}
+
+// Sets every bit of byte k when byte k of v is a quotable character ('"',
+// '\\' or a control character).
+simdjson_inline escape_vector escape_flags(escape_vector v) noexcept {
+  const __m128i v34 = _mm_set1_epi8(34); // '"'
+  const __m128i v92 = _mm_set1_epi8(92); // '\\'
+  const __m128i v31 = _mm_set1_epi8(31); // for control char detection
+  __m128i needs_escape = _mm_cmpeq_epi8(v, v34);
+  needs_escape = _mm_or_si128(needs_escape, _mm_cmpeq_epi8(v, v92));
+  return _mm_or_si128(
+      needs_escape, _mm_cmpeq_epi8(_mm_subs_epu8(v, v31), _mm_setzero_si128()));
+}
+
+// True when any byte needs escaping. Kept separate from escape_bitmask
+// because some instruction sets can answer it without leaving the vector
+// register file; on SSE2 the compiler folds the two together.
+simdjson_inline bool escape_any(escape_vector flags) noexcept {
+  return _mm_movemask_epi8(flags) != 0;
+}
+
+// A 16-bit mask with bit k set when byte k needed escaping.
+simdjson_inline uint64_t escape_bitmask(escape_vector flags) noexcept {
+  return uint64_t(uint32_t(_mm_movemask_epi8(flags)));
+}
+
+#else // SIMDJSON_EXPERIMENTAL_HAS_NEON
+
+using escape_vector = uint8x16_t;
+static constexpr unsigned escape_mask_bits = 4;
+
+simdjson_inline escape_vector escape_load16(const uint8_t *p) noexcept {
+  return vld1q_u8(p);
+}
+
+simdjson_inline void escape_store16(char *out, escape_vector v) noexcept {
+  vst1q_u8(reinterpret_cast<uint8_t *>(out), v);
+}
+
+simdjson_inline escape_vector escape_load8x2(const uint8_t *a,
+                                             const uint8_t *b) noexcept {
+  uint64_t a64, b64;
+  memcpy(&a64, a, 8);
+  memcpy(&b64, b, 8);
+  return vreinterpretq_u8_u64(vsetq_lane_u64(b64, vdupq_n_u64(a64), 1));
+}
+
+simdjson_inline escape_vector escape_load4x2(const uint8_t *a,
+                                             const uint8_t *b) noexcept {
+  uint32_t a32, b32;
+  memcpy(&a32, a, 4);
+  memcpy(&b32, b, 4);
+  return vreinterpretq_u8_u32(vsetq_lane_u32(b32, vdupq_n_u32(a32), 1));
+}
+
+simdjson_inline escape_vector escape_flags(escape_vector v) noexcept {
+  uint8x16_t needs_escape = vceqq_u8(v, vdupq_n_u8(34));              // '"'
+  needs_escape = vorrq_u8(needs_escape, vceqq_u8(v, vdupq_n_u8(92))); // '\\'
+  return vorrq_u8(needs_escape, vcltq_u8(v, vdupq_n_u8(32)));
+}
+
+simdjson_inline bool escape_any(escape_vector flags) noexcept {
+  return vmaxvq_u32(vreinterpretq_u32_u8(flags)) != 0;
+}
+
+// Four bits per byte rather than one: escape_block and the tail paths scale
+// their shifts by escape_mask_bits to match.
+simdjson_inline uint64_t escape_bitmask(escape_vector flags) noexcept {
+  uint8x8_t narrowed = vshrn_n_u16(vreinterpretq_u16_u8(flags), 4);
+  return vget_lane_u64(vreinterpret_u64_u8(narrowed), 0);
+}
+
+#endif // instruction set selection
+
+// The escape bitmask of a 16-byte block.
+simdjson_inline uint64_t escape_mask(escape_vector v) noexcept {
+  return escape_bitmask(escape_flags(v));
+}
+
+// Copies n bytes with n < 16, using overlapping loads and stores. It never
+// reads more than n bytes from src, nor writes more than n bytes to dst.
+simdjson_inline void copy_lt16(char *dst, const uint8_t *src,
+                               size_t n) noexcept {
+  if (n >= 8) {
+    memcpy(dst, src, 8);
+    memcpy(dst + n - 8, src + n - 8, 8);
+  } else if (n >= 4) {
+    memcpy(dst, src, 4);
+    memcpy(dst + n - 4, src + n - 4, 4);
+  } else if (n > 0) {
+    dst[0] = char(src[0]);
+    dst[n >> 1] = char(src[n >> 1]);
+    dst[n - 1] = char(src[n - 1]);
+  }
+}
+
+// Escapes the bytes of src in the range [i, blockend), given that m is the
+// (non-zero) escape mask for that range: the escape_mask_bits-wide lane of m
+// at byte k is non-zero when src[i + k] requires escaping. Returns the updated
+// output pointer.
+simdjson_never_inline char *escape_block(const uint8_t *src, char *out,
+                                         size_t i, size_t blockend,
+                                         uint64_t m) noexcept {
+  constexpr uint64_t lane = (uint64_t(1) << escape_mask_bits) - 1;
+
+  size_t pos = i; // first byte not yet copied
+  while (m) {
+    const size_t tz = trailing_zeroes(m);
+    const size_t next = i + tz / escape_mask_bits;
+    // Copy the run of safe bytes that precedes this escape.
+    copy_lt16(out, src + pos, next - pos);
+    out += next - pos;
+    escape_json_char(char(src[next]), out);
+    pos = next + 1;
+    m &= ~(lane << tz);
+  }
+  // Copy whatever follows the last escape.
+  copy_lt16(out, src + pos, blockend - pos);
+  return out + (blockend - pos);
+}
+
+// Writes the escaped version of input to out, returning the number of bytes
+// written.
+simdjson_really_inline size_t write_string_escaped(const std::string_view input, char *out) {
+  const size_t len = input.size();
+  const uint8_t *src = reinterpret_cast<const uint8_t *>(input.data());
+  const char *const initout = out;
+
+  size_t i = 0;
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2 && defined(__AVX2__)
+  while (i + 32 <= len) {
+    const __m256i word = _mm256_loadu_si256(reinterpret_cast<const __m256i *>(src + i));
+    const __m256i flags = _mm256_or_si256(
+        _mm256_or_si256(_mm256_cmpeq_epi8(word, _mm256_set1_epi8(34)),   // '"'
+                        _mm256_cmpeq_epi8(word, _mm256_set1_epi8(92))),  // '\\'
+        _mm256_cmpeq_epi8(_mm256_subs_epu8(word, _mm256_set1_epi8(31)),
+                          _mm256_setzero_si256()));                      // control
+    const uint32_t mask = uint32_t(_mm256_movemask_epi8(flags));
+    if (simdjson_likely(mask == 0)) {
+      _mm256_storeu_si256(reinterpret_cast<__m256i *>(out), word);
+      out += 32;
+    } else {
+      for (size_t half = 0; half < 32; half += 16) {
+        const uint64_t m = (mask >> half) & 0xFFFF;
+        if (m == 0) {
+          escape_store16(out, escape_load16(src + i + half));
+          out += 16;
+        } else {
+          out = escape_block(src, out, i + half, i + half + 16, m);
+        }
+      }
+    }
+    i += 32;
+  }
+#endif
+  while (i + 16 <= len) {
+    escape_vector word = escape_load16(src + i);
+    escape_vector flags = escape_flags(word);
+    if (simdjson_likely(!escape_any(flags))) {
+      escape_store16(out, word);
+      out += 16;
+    } else {
+      out = escape_block(src, out, i, i + 16, escape_bitmask(flags));
+    }
+    i += 16;
+  }
+  if (i < len) {
+    const size_t rem = len - i;
+    uint64_t m;
+    if (len >= 16) {
+      // The last 16 bytes of the input are in bounds. Bit k of that block's
+      // mask belongs to input position len - 16 + k, so shift it down to align
+      // bit 0 with position i.
+      m = escape_mask(escape_load16(src + len - 16)) >>
+          (escape_mask_bits * (16 - rem));
+    } else if (len >= 8) {
+      // Here i == 0 and rem == len. Two overlapping 8-byte loads cover
+      // [0, 8) and [len - 8, len), which is the whole input since len < 16.
+      uint64_t mm = escape_mask(escape_load8x2(src, src + len - 8));
+      constexpr uint64_t low8 = (uint64_t(1) << (escape_mask_bits * 8)) - 1;
+      m = (mm & low8) |
+          ((mm >> (escape_mask_bits * 8)) << (escape_mask_bits * (len - 8)));
+    } else if (len >= 4) {
+      // Same idea with two overlapping 4-byte loads.
+      uint64_t mm = escape_mask(escape_load4x2(src, src + len - 4));
+      constexpr uint64_t low4 = (uint64_t(1) << (escape_mask_bits * 4)) - 1;
+      m = (mm & low4) | (((mm >> (escape_mask_bits * 4)) & low4)
+                         << (escape_mask_bits * (len - 4)));
+    } else {
+      // Fewer than 4 bytes: at most three table lookups, no need for SIMD.
+      for (size_t k = 0; k < len; k++) {
+        uint8_t c = src[k];
+        if (json_quotable_character[c]) {
+          escape_json_char(char(c), out);
+        } else {
+          *out++ = char(c);
+        }
+      }
+      return size_t(out - initout);
+    }
+    if (m == 0) {
+      copy_lt16(out, src + i, rem);
+      out += rem;
+    } else {
+      out = escape_block(src, out, i, len, m);
+    }
+  }
+  return size_t(out - initout);
+}
+
+#else // SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
 // Writes the escaped version of input to out, returning the number of bytes
 // written. Uses SIMD position finding to locate quotable characters efficiently.
 inline size_t write_string_escaped(const std::string_view input, char *out) {
@@ -40357,9 +48393,13 @@ inline size_t write_string_escaped(const std::string_view input, char *out) {
     escape_json_char(input[location], out);
     location += 1;
   }
-  return out - initout;
+  return size_t(out - initout);
 }

+#endif // SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+#undef SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
+
 simdjson_inline string_builder::string_builder(size_t initial_capacity)
     : buffer(new(std::nothrow) char[initial_capacity]), position(0),
       capacity(buffer.get() != nullptr ? initial_capacity : 0),
@@ -40382,7 +48422,7 @@ simdjson_inline bool string_builder::capacity_check(size_t upcoming_bytes) {
   return is_valid;
 }

-simdjson_inline void string_builder::grow_buffer(size_t desired_capacity) {
+inline void string_builder::grow_buffer(size_t desired_capacity) {
   if (!is_valid) {
     return;
   }
@@ -40438,81 +48478,136 @@ simdjson_inline void string_builder::clear() noexcept {

 namespace internal {

-template <typename number_type, typename = typename std::enable_if<
-                                    std::is_unsigned<number_type>::value>::type>
-simdjson_really_inline int int_log2(number_type x) {
-  return 63 - leading_zeroes(uint64_t(x) | 1);
-}
-
-simdjson_really_inline int fast_digit_count_32(uint32_t x) {
-  static uint64_t table[] = {
-      4294967296,  8589934582,  8589934582,  8589934582,  12884901788,
-      12884901788, 12884901788, 17179868184, 17179868184, 17179868184,
-      21474826480, 21474826480, 21474826480, 21474826480, 25769703776,
-      25769703776, 25769703776, 30063771072, 30063771072, 30063771072,
-      34349738368, 34349738368, 34349738368, 34349738368, 38554705664,
-      38554705664, 38554705664, 41949672960, 41949672960, 41949672960,
-      42949672960, 42949672960};
-  return uint32_t((x + table[int_log2(x)]) >> 32);
-}
-
-simdjson_really_inline int fast_digit_count_64(uint64_t x) {
-  static uint64_t table[] = {9,
-                             99,
-                             999,
-                             9999,
-                             99999,
-                             999999,
-                             9999999,
-                             99999999,
-                             999999999,
-                             9999999999,
-                             99999999999,
-                             999999999999,
-                             9999999999999,
-                             99999999999999,
-                             999999999999999ULL,
-                             9999999999999999ULL,
-                             99999999999999999ULL,
-                             999999999999999999ULL,
-                             9999999999999999999ULL};
-  int y = (19 * int_log2(x) >> 6);
-  y += x > table[y];
-  return y + 1;
-}
-
-template <typename number_type, typename = typename std::enable_if<
-                                    std::is_unsigned<number_type>::value>::type>
-simdjson_really_inline size_t digit_count(number_type v) noexcept {
-  static_assert(sizeof(number_type) == 8 || sizeof(number_type) == 4 ||
-                    sizeof(number_type) == 2 || sizeof(number_type) == 1,
-                "We only support 8-bit, 16-bit, 32-bit and 64-bit numbers");
-  SIMDJSON_IF_CONSTEXPR(sizeof(number_type) <= 4) {
-    return fast_digit_count_32(static_cast<uint32_t>(v));
+// Integer to decimal: James Edward Anhalt III's algorithm
+static const char jeaiii_dd[201] =
+    "00010203040506070809101112131415161718192021222324252627282930313233343536373839"
+    "40414243444546474849505152535455565758596061626364656667686970717273747576777879"
+    "8081828384858687888990919293949596979899";
+static const char jeaiii_fd[201] =
+    "0\0" "1\0" "2\0" "3\0" "4\0" "5\0" "6\0" "7\0" "8\0" "9\0"
+    "10111213141516171819202122232425262728293031323334353637383940414243444546474849"
+    "50515253545556575859606162636465666768697071727374757677787980818283848586878889"
+    "90919293949596979899";
+
+simdjson_really_inline void jeaiii_write_dd(char *p, uint64_t k) noexcept {
+  std::memcpy(p, &jeaiii_dd[2 * k], 2);
+}
+simdjson_really_inline void jeaiii_write_fd(char *p, uint64_t k) noexcept {
+  std::memcpy(p, &jeaiii_fd[2 * k], 2);
+}
+
+// Caller guarantees n < 10^8. Writes 1 to 8 digits.
+simdjson_really_inline char *jeaiii_lt1e8(char *b, uint32_t n) noexcept {
+  constexpr uint64_t mask24 = (uint64_t(1) << 24) - 1;
+  constexpr uint64_t mask32 = (uint64_t(1) << 32) - 1;
+  if (n < 100) {
+    jeaiii_write_fd(b, n);
+    return n < 10 ? b + 1 : b + 2;
+  }
+  if (n < 1000000) {
+    if (n < 10000) {
+      const uint32_t f0 = uint32_t(10 * (1 << 24) / 1e3 + 1) * n;
+      jeaiii_write_fd(b, f0 >> 24);
+      b -= n < 1000;
+      const uint32_t f2 = uint32_t(f0 & mask24) * 100;
+      jeaiii_write_dd(b + 2, f2 >> 24);
+      return b + 4;
+    }
+    const uint64_t f0 = uint64_t(10 * (1ull << 32) / 1e5 + 1) * n;
+    jeaiii_write_fd(b, f0 >> 32);
+    b -= n < 100000;
+    const uint64_t f2 = (f0 & mask32) * 100;
+    jeaiii_write_dd(b + 2, f2 >> 32);
+    const uint64_t f4 = (f2 & mask32) * 100;
+    jeaiii_write_dd(b + 4, f4 >> 32);
+    return b + 6;
+  }
+  const uint64_t f0 = uint64_t(10 * (1ull << 48) / 1e7 + 1) * n >> 16;
+  jeaiii_write_fd(b, f0 >> 32);
+  b -= n < 10000000;
+  const uint64_t f2 = (f0 & mask32) * 100;
+  jeaiii_write_dd(b + 2, f2 >> 32);
+  const uint64_t f4 = (f2 & mask32) * 100;
+  jeaiii_write_dd(b + 4, f4 >> 32);
+  const uint64_t f6 = (f4 & mask32) * 100;
+  jeaiii_write_dd(b + 6, f6 >> 32);
+  return b + 8;
+}
+
+// Caller guarantees z < 10^8. Always writes exactly 8 digits.
+simdjson_really_inline char *jeaiii_8_digits(char *b, uint32_t z) noexcept {
+  constexpr uint64_t mask32 = (uint64_t(1) << 32) - 1;
+  const uint64_t f0 = (uint64_t((1ull << 48) / 1e6 + 1) * z >> 16) + 1;
+  jeaiii_write_dd(b, f0 >> 32);
+  const uint64_t f2 = (f0 & mask32) * 100;
+  jeaiii_write_dd(b + 2, f2 >> 32);
+  const uint64_t f4 = (f2 & mask32) * 100;
+  jeaiii_write_dd(b + 4, f4 >> 32);
+  const uint64_t f6 = (f4 & mask32) * 100;
+  jeaiii_write_dd(b + 6, f6 >> 32);
+  return b + 8;
+}
+
+// Caller guarantees 10^8 <= n < 2^32. Writes 9 or 10 digits.
+simdjson_really_inline char *jeaiii_9_or_10(char *b, uint64_t n) noexcept {
+  constexpr uint64_t mask57 = (uint64_t(1) << 57) - 1;
+  const uint64_t f0 = uint64_t(10 * (1ull << 57) / 1e9 + 1) * n;
+  jeaiii_write_fd(b, f0 >> 57);
+  b -= n < 1000000000;
+  const uint64_t f2 = (f0 & mask57) * 100;
+  jeaiii_write_dd(b + 2, f2 >> 57);
+  const uint64_t f4 = (f2 & mask57) * 100;
+  jeaiii_write_dd(b + 4, f4 >> 57);
+  const uint64_t f6 = (f4 & mask57) * 100;
+  jeaiii_write_dd(b + 6, f6 >> 57);
+  const uint64_t f8 = (f6 & mask57) * 100;
+  jeaiii_write_dd(b + 8, f8 >> 57);
+  return b + 10;
+}
+
+simdjson_really_inline char *write_uint_jeaiii(char *b, uint64_t n) noexcept {
+  if (n < 100000000) {
+    return jeaiii_lt1e8(b, uint32_t(n));
+  }
+  if (n < (uint64_t(1) << 32)) {
+    return jeaiii_9_or_10(b, n);
+  }
+  // At least 10 digits: the low 8 digits, and 2 to 12 digits above them.
+  const uint32_t z = uint32_t(n % 100000000);
+  uint64_t u = n / 100000000;
+  if (u < 100000000) {
+    // u has 2 to 8 digits (if u < 10, n would be below 2^32).
+    b = jeaiii_lt1e8(b, uint32_t(u));
+  } else if (u < (uint64_t(1) << 32)) {
+    b = jeaiii_9_or_10(b, u);
+  } else {
+    // u has 11 or 12 digits: split off 8 more.
+    const uint32_t y = uint32_t(u % 100000000);
+    u /= 100000000;
+    b = jeaiii_lt1e8(b, uint32_t(u)); // 3 or 4 digits
+    b = jeaiii_8_digits(b, y);
   }
-  else {
-    return fast_digit_count_64(static_cast<uint64_t>(v));
-  }
-}
-static const char decimal_table[200] = {
-    0x30, 0x30, 0x30, 0x31, 0x30, 0x32, 0x30, 0x33, 0x30, 0x34, 0x30, 0x35,
-    0x30, 0x36, 0x30, 0x37, 0x30, 0x38, 0x30, 0x39, 0x31, 0x30, 0x31, 0x31,
-    0x31, 0x32, 0x31, 0x33, 0x31, 0x34, 0x31, 0x35, 0x31, 0x36, 0x31, 0x37,
-    0x31, 0x38, 0x31, 0x39, 0x32, 0x30, 0x32, 0x31, 0x32, 0x32, 0x32, 0x33,
-    0x32, 0x34, 0x32, 0x35, 0x32, 0x36, 0x32, 0x37, 0x32, 0x38, 0x32, 0x39,
-    0x33, 0x30, 0x33, 0x31, 0x33, 0x32, 0x33, 0x33, 0x33, 0x34, 0x33, 0x35,
-    0x33, 0x36, 0x33, 0x37, 0x33, 0x38, 0x33, 0x39, 0x34, 0x30, 0x34, 0x31,
-    0x34, 0x32, 0x34, 0x33, 0x34, 0x34, 0x34, 0x35, 0x34, 0x36, 0x34, 0x37,
-    0x34, 0x38, 0x34, 0x39, 0x35, 0x30, 0x35, 0x31, 0x35, 0x32, 0x35, 0x33,
-    0x35, 0x34, 0x35, 0x35, 0x35, 0x36, 0x35, 0x37, 0x35, 0x38, 0x35, 0x39,
-    0x36, 0x30, 0x36, 0x31, 0x36, 0x32, 0x36, 0x33, 0x36, 0x34, 0x36, 0x35,
-    0x36, 0x36, 0x36, 0x37, 0x36, 0x38, 0x36, 0x39, 0x37, 0x30, 0x37, 0x31,
-    0x37, 0x32, 0x37, 0x33, 0x37, 0x34, 0x37, 0x35, 0x37, 0x36, 0x37, 0x37,
-    0x37, 0x38, 0x37, 0x39, 0x38, 0x30, 0x38, 0x31, 0x38, 0x32, 0x38, 0x33,
-    0x38, 0x34, 0x38, 0x35, 0x38, 0x36, 0x38, 0x37, 0x38, 0x38, 0x38, 0x39,
-    0x39, 0x30, 0x39, 0x31, 0x39, 0x32, 0x39, 0x33, 0x39, 0x34, 0x39, 0x35,
-    0x39, 0x36, 0x39, 0x37, 0x39, 0x38, 0x39, 0x39,
-};
+  return jeaiii_8_digits(b, z);
+}
+
+// Writes v at p, which must have to_chars_buffer_size bytes available, and
+// returns the end of what was written.
+simdjson_inline char *write_double(char *p, double v) noexcept {
+#if SIMDJSON_ENABLE_NAN_INF
+  if (simdjson_unlikely(!std::isfinite(v))) {
+    if (std::isnan(v)) {
+      std::memcpy(p, "NaN", 3);
+      return p + 3;
+    }
+    if (v < 0) {
+      *p++ = '-';
+    }
+    std::memcpy(p, "Infinity", 8);
+    return p + 8;
+  }
+#endif
+  return simdjson::internal::to_chars(p, nullptr, v);
+}
 } // namespace internal

 template <typename number_type, typename>
@@ -40540,87 +48635,62 @@ simdjson_inline void string_builder::append(number_type v) noexcept {
     }
   }
   else SIMDJSON_IF_CONSTEXPR(std::is_unsigned<number_type>::value) {
-    // Process 4 digits at a time instead of 2, reducing store operations
-    // and divisions by approximately half for large numbers.
     constexpr size_t max_number_size = 20;
     if (capacity_check(max_number_size)) {
       using unsigned_type = typename std::make_unsigned<number_type>::type;
-      unsigned_type pv = static_cast<unsigned_type>(v);
-      size_t dc = internal::digit_count(pv);
-      char *write_pointer = buffer.get() + position + dc - 1;
-
-      // Process 4 digits per iteration for large numbers
-      while (pv >= 10000) {
-        unsigned_type q = pv / 10000;
-        unsigned_type r = pv % 10000;
-        unsigned_type r_hi = r / 100;  // High 2 digits of remainder
-        unsigned_type r_lo = r % 100;  // Low 2 digits of remainder
-        // Write low 2 digits first (rightmost), then high 2 digits
-        memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
-        memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
-        write_pointer -= 4;
-        pv = q;
-      }
-
-      // Handle remaining 1-4 digits with original 2-digit loop
-      while (pv >= 100) {
-        memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
-        write_pointer -= 2;
-        pv /= 100;
-      }
-      if (pv >= 10) {
-        *write_pointer-- = char('0' + (pv % 10));
-        pv /= 10;
-      }
-      *write_pointer = char('0' + pv);
-      position += dc;
+      char* end = internal::write_uint_jeaiii(
+          buffer.get() + position,
+          static_cast<uint64_t>(static_cast<unsigned_type>(v)));
+      position = end - buffer.get();
     }
   }
   else SIMDJSON_IF_CONSTEXPR(std::is_integral<number_type>::value) {
-    // Same 4-digit batching as unsigned path for signed integers
+    // 19 digits (max abs value of int64_t) + optional minus sign.
     constexpr size_t max_number_size = 20;
     if (capacity_check(max_number_size)) {
       using unsigned_type = typename std::make_unsigned<number_type>::type;
       bool negative = v < 0;
-      unsigned_type pv = static_cast<unsigned_type>(v);
-      if (negative) {
-        pv = 0 - pv; // the 0 is for Microsoft
-      }
-      size_t dc = internal::digit_count(pv);
-      // by always writing the minus sign, we avoid the branch.
+      // 0 - pv (rather than -pv) avoids an MSVC unary-minus warning.
+      unsigned_type pv = negative
+          ? unsigned_type(0) - static_cast<unsigned_type>(v)
+          : static_cast<unsigned_type>(v);
+      // Branchless: always write '-', advance only if negative.
       buffer.get()[position] = '-';
-      position += negative ? 1 : 0;
-      char *write_pointer = buffer.get() + position + dc - 1;
-
-      // Process 4 digits per iteration for large numbers
-      while (pv >= 10000) {
-        unsigned_type q = pv / 10000;
-        unsigned_type r = pv % 10000;
-        unsigned_type r_hi = r / 100;
-        unsigned_type r_lo = r % 100;
-        memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
-        memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
-        write_pointer -= 4;
-        pv = q;
-      }
-
-      // Handle remaining 1-4 digits
-      while (pv >= 100) {
-        memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
-        write_pointer -= 2;
-        pv /= 100;
-      }
-      if (pv >= 10) {
-        *write_pointer-- = char('0' + (pv % 10));
-        pv /= 10;
-      }
-      *write_pointer = char('0' + pv);
-      position += dc;
+      position += negative;
+      char* end = internal::write_uint_jeaiii(
+          buffer.get() + position, static_cast<uint64_t>(pv));
+      position = end - buffer.get();
     }
   }
   else SIMDJSON_IF_CONSTEXPR(std::is_floating_point<number_type>::value) {
-    constexpr size_t max_number_size = 24;
+    // Must reserve to_chars_buffer_size (40): only ~24 chars are emitted,
+    // but to_chars over-writes with fixed-size 16/17-byte copies so the
+    // compiler can inline mem* (see simdjson::internal::to_chars_buffer_size).
+    constexpr size_t max_number_size = simdjson::internal::to_chars_buffer_size;
     if (capacity_check(max_number_size)) {
+#if SIMDJSON_ENABLE_NAN_INF
+      // Check if the input might be NaN or infinity
+      if (simdjson_unlikely(!std::isfinite(v))) {
+        if (std::isnan(v)) {
+          constexpr char nan_literal[] = "NaN";
+          constexpr size_t nan_len = sizeof(nan_literal) - 1;
+
+          std::memcpy(buffer.get() + position, nan_literal, nan_len);
+          position += nan_len;
+        } else {
+          constexpr char inf_literal[] = "Infinity";
+          constexpr size_t inf_len = sizeof(inf_literal) - 1;
+          if (v < 0) {
+            buffer.get()[position] = '-';
+            ++position;
+          }
+          std::memcpy(buffer.get() + position, inf_literal, inf_len);
+          position += inf_len;
+        }
+        return;
+      }
+#endif
+
       // We could specialize for float.
       char *end = simdjson::internal::to_chars(buffer.get() + position, nullptr,
                                                double(v));
@@ -40681,7 +48751,9 @@ simdjson_inline void string_builder::escape_and_append_with_quotes() noexcept {
 #endif

 simdjson_inline void string_builder::append_raw(const char *c) noexcept {
-  size_t len = std::strlen(c);
+  // char_traits::length is constexpr; lets the compiler fold the length
+  // when called with a pointer to a compile-time-constant string.
+  size_t len = std::char_traits<char>::length(c);
   append_raw(c, len);
 }

@@ -40700,6 +48772,14 @@ simdjson_inline void string_builder::append_raw(const char *str,
     position += len;
   }
 }
+
+template <size_t N>
+simdjson_inline void string_builder::append_raw_n(const char *str) noexcept {
+  if (capacity_check(N)) {
+    std::memcpy(buffer.get() + position, str, N);
+    position += N;
+  }
+}
 #if SIMDJSON_SUPPORTS_CONCEPTS
 // Support for optional types (std::optional, etc.)
 template <concepts::optional_type T>
@@ -40729,7 +48809,7 @@ simdjson_inline void string_builder::append(const T &value) {
 #if SIMDJSON_SUPPORTS_RANGES && SIMDJSON_SUPPORTS_CONCEPTS
 // Support for range-based appending (std::ranges::view, etc.)
 template <std::ranges::range R>
-  requires(!std::is_convertible<R, std::string_view>::value && !require_custom_serialization<R>)
+  requires(!std::is_convertible<R, std::string_view>::value && !concepts::optional_type<R> && !require_custom_serialization<R>)
 simdjson_inline void string_builder::append(const R &range) noexcept {
   auto it = std::ranges::begin(range);
   auto end = std::ranges::end(range);
@@ -41335,7 +49415,7 @@ public:
 #if SIMDJSON_SUPPORTS_RANGES && SIMDJSON_SUPPORTS_CONCEPTS
   // Support for range-based appending (std::ranges::view, etc.)
   template <std::ranges::range R>
-requires (!std::is_convertible<R, std::string_view>::value && !require_custom_serialization<R>)
+requires (!std::is_convertible<R, std::string_view>::value && !concepts::optional_type<R> && !require_custom_serialization<R>)
   simdjson_inline void append(const R &range) noexcept;
 #endif
   /**
@@ -41349,6 +49429,15 @@ requires (!std::is_convertible<R, std::string_view>::value && !require_custom_se
    * There is no UTF-8 validation.
    */
   simdjson_inline void append_raw(const char *str, size_t len) noexcept;
+
+  /**
+   * Append exactly N characters from str. The length is a template parameter
+   * so the compiler can fully inline the memcpy with a compile-time-constant
+   * size, avoiding the libc call. Used for compile-time-constant keys in the
+   * reflection struct atom.
+   */
+  template <size_t N>
+  simdjson_inline void append_raw_n(const char *str) noexcept;
 #if SIMDJSON_EXCEPTIONS
   /**
    * Creates an std::string from the written JSON buffer.
@@ -41400,6 +49489,26 @@ requires (!std::is_convertible<R, std::string_view>::value && !require_custom_se
    */
   simdjson_inline size_t size() const noexcept;

+  // ============================================================
+  // Internal hooks for the position-as-local writer in json_builder.h.
+  // These exist so the reflection atom code can hold buffer pointer,
+  // position and capacity in registers across long write chains rather
+  // than reloading them after every char* write (strict aliasing
+  // forces those reloads when accessed via members of *this). User
+  // code should NOT call these directly.
+  // ============================================================
+  simdjson_inline char *unsafe_data() noexcept { return buffer.get(); }
+  simdjson_inline size_t unsafe_position() const noexcept { return position; }
+  simdjson_inline size_t unsafe_capacity() const noexcept { return capacity; }
+  simdjson_inline void unsafe_set_position(size_t p) noexcept { position = p; }
+  /// Make capacity available for at least `n` more bytes after the current
+  /// position. Returns false if the allocation failed.
+  simdjson_inline bool unsafe_grow(size_t needed_total_capacity) noexcept {
+    grow_buffer(needed_total_capacity);
+    return is_valid;
+  }
+  simdjson_inline bool unsafe_is_valid() const noexcept { return is_valid; }
+
 private:
   /**
    * Returns true if we can write at least upcoming_bytes bytes.
@@ -41413,7 +49522,7 @@ private:
    * If the allocation fails, is_valid is set to false. We expect
    * that this function would not be repeatedly called.
    */
-  simdjson_inline void grow_buffer(size_t desired_capacity);
+  inline void grow_buffer(size_t desired_capacity);

   /**
    * We use this helper function to make sure that is_valid is kept consistent.
@@ -41470,6 +49579,7 @@ simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initi
 /* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_STRING_BUILDER_H */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/builder/json_string_builder.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/concepts.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/annotations.h" */
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
 #if SIMDJSON_STATIC_REFLECTION

@@ -41487,64 +49597,370 @@ namespace simdjson {
 namespace fallback {
 namespace builder {

-template <class T>
+// Forward-declare helpers defined in json_string_builder-inl.h so the
+// writer-based atom code below can call them (the -inl.h is not yet
+// included at the point this header is parsed; without these forwards,
+// name lookup falls back to the wrong outer namespace).
+namespace internal {
+simdjson_really_inline char *write_uint_jeaiii(char *p, uint64_t v) noexcept;
+simdjson_inline char *write_double(char *p, double v) noexcept;
+} // namespace internal
+simdjson_really_inline size_t write_string_escaped(const std::string_view input, char *out);
+
+SIMDJSON_PUSH_DISABLE_WARNINGS
+SIMDJSON_DISABLE_GCC_WARNING(-Warray-bounds)
+#if !defined(__clang__)
+SIMDJSON_DISABLE_GCC_WARNING(-Wstringop-overflow)
+#endif
+
+// =============================================================
+// `writer`: position-as-local hot-path writer used by the reflection
+// atom code below. Holds the buffer pointer, write position and
+// capacity in three fields that, once `writer` itself is a stack-local
+// in the caller and all atom() functions are inlined, become true
+// register-resident locals after SROA. Glaze achieves the same effect
+// by passing `B&& b, auto&& ix` through every helper. Holding `pos`
+// in a register (rather than as a member of string_builder) is what
+// breaks the strict-aliasing penalty on every char* write through the
+// buffer, which forces a reload of `b.position` and `b.capacity`
+// after every byte.
+//
+// basic_writer<false> (below) is the unchecked variant: the caller has
+// already reserved enough capacity for everything the write chain can
+// produce (see bound_detail::size_bound), so ensure() compiles away.
+// =============================================================
+template <bool Checked>
+struct basic_writer {
+  static constexpr bool checked = Checked;
+  char *ptr;        // buffer pointer (refreshed after a grow)
+  size_t pos;       // write position (local)
+  size_t cap;       // capacity (refreshed after a grow)
+  string_builder &sb;  // back-ref for grow / sync
+
+  // Snapshot string_builder state into a writer for the duration of
+  // a write chain.
+  simdjson_really_inline basic_writer(string_builder &builder) noexcept
+      : ptr(builder.unsafe_data())
+      , pos(builder.unsafe_position())
+      , cap(builder.unsafe_capacity())
+      , sb(builder) {}
+
+  // Write the local position back to the underlying string_builder.
+  // Caller is responsible for invoking before the writer is dropped
+  // (otherwise data is lost). Idempotent.
+  simdjson_really_inline void sync() noexcept {
+    sb.unsafe_set_position(pos);
+  }
+
+  // Ensure at least `n` more bytes of free capacity. Grows the
+  // underlying buffer if needed (rare path). Returns false on
+  // allocation failure.
+  simdjson_really_inline bool ensure(size_t n) noexcept {
+    // pos <= cap, and cap is the size of a live allocation, so pos + n
+    // cannot wrap when n is a small constant or a compile-time length.
+    // Callers passing a size derived from input (the string atoms) must
+    // bound it against pos themselves. Keep the `pos + n <= cap` form:
+    // `n <= cap - pos` is measurably slower once the serializer is inlined.
+    if (simdjson_likely(pos + n <= cap)) { return true; }
+    return grow_slow(n);
+  }
+
+  simdjson_never_inline bool grow_slow(size_t n) noexcept {
+    // Detect overflow.
+    // This is pedantic except maybe on 32-bit targets.
+    if (simdjson_unlikely(pos + n < pos)) return false;
+    sb.unsafe_set_position(pos);
+    // even if 2*capacity overflows, the (std::max) below will pick the needed value,
+    // so we do not need a separate overflow check here.
+    if (!sb.unsafe_grow((std::max)(cap * 2, pos + n))) {
+      // The string_builder freed its buffer and is now invalid (null buffer,
+      // zero capacity and position). Mirror that state so that every later
+      // ensure() fails too: callers only return from the current atom, and
+      // their callers keep writing.
+      ptr = nullptr;
+      pos = 0;
+      cap = 0;
+      return false;
+    }
+    ptr = sb.unsafe_data();
+    cap = sb.unsafe_capacity();
+    return true;
+  }
+};
+
+// The unchecked writer writes into a raw buffer that the caller sized with
+// serialized_size_bound: it never grows and needs no string_builder.
+template <>
+struct basic_writer<false> {
+  static constexpr bool checked = false;
+  char *ptr;
+  size_t pos;
+
+  simdjson_really_inline basic_writer(char *buffer, size_t position) noexcept
+      : ptr(buffer), pos(position) {}
+
+  simdjson_really_inline bool ensure(size_t) const noexcept { return true; }
+};
+
+using writer = basic_writer<true>;
+using unchecked_writer = basic_writer<false>;
+
+// Bytes reserved past the size bound for an unchecked writer: it may then
+// write a little past the end of what it produces (e.g., copy keys as whole
+// 16-byte blocks).
+inline constexpr size_t unchecked_slack = 64;
+
+consteval size_t padded_key_length(size_t length) {
+  return (length + 15) / 16 * 16;
+}
+
+// === Helper: invoke a string_builder member that writes variable-length
+// content (escape_and_append_with_quotes etc), syncing the writer's local
+// state before the call and reloading after. Used for string fields where
+// rewriting the entire SIMD escape path through the writer would be a much
+// bigger refactor. f may be user code (a with<Adapter> serializer) that
+// throws: the exception then propagates to the caller.
+template <class W, class F>
+simdjson_really_inline void call_through_string_builder(W &w, F &&f) noexcept(noexcept(f(w.sb))) {
+  w.sync();
+  f(w.sb);
+  w.ptr = w.sb.unsafe_data();
+  w.pos = w.sb.unsafe_position();
+  w.cap = w.sb.unsafe_capacity();
+}
+
+// Helpers implementing the serialization side of the annotations (see
+// simdjson/annotations.h) for reflected structures.
+namespace annotation_detail {
+
+// A member is serialized unless it is annotated with skip or skip_serializing.
+consteval bool is_serialized_member(std::meta::info dm) {
+  return !simdjson::detail::has_annotation(dm, ^^simdjson::detail::skip_tag)
+      && !simdjson::detail::has_annotation(dm, ^^simdjson::detail::skip_serializing_tag);
+}
+
+// False when the member has a skip_serializing_if<pred> annotation and
+// pred(value) is true.
+template <auto dm, typename V>
+simdjson_really_inline bool should_serialize(const V &value) {
+  constexpr std::meta::info skip_if_type = simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::skip_serializing_if_t);
+  if constexpr (skip_if_type != std::meta::info{}) {
+    using skip_if = typename [: skip_if_type :];
+    return !skip_if::predicate(value);
+  } else {
+    (void)value;
+    return true;
+  }
+}
+
+// Serialize a member value, through its with<Adapter> annotation when the
+// adapter provides a serialize function.
+template <auto dm, class W, typename V>
+simdjson_really_inline void atom_member(W &w, const V &value) {
+  constexpr std::meta::info with_type = simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::with_t);
+  if constexpr (with_type != std::meta::info{}) {
+    using adapter = typename [: with_type :]::adapter;
+    if constexpr (requires(string_builder &b) { adapter::serialize(b, value); }) {
+      call_through_string_builder(w, [&](string_builder &b) { adapter::serialize(b, value); });
+    } else {
+      atom(w, value);
+    }
+  } else {
+    atom(w, value);
+  }
+}
+
+// Write the "key":value pairs of the members of t (without the braces), each
+// preceded by a comma unless it is the first one. The members of a member
+// annotated with flatten are written in its place.
+template <class W, class T>
+simdjson_really_inline void atom_fields(W &w, const T &t, bool &first) {
+  // Per-field block: ensure key+value worst case, then write key + value
+  // through the writer's local pos. For arithmetic fields, the integer
+  // write happens directly via write_uint_jeaiii on w.ptr+w.pos, so pos
+  // never round-trips through memory.
+  template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+    if constexpr (is_serialized_member(dm)) {
+      if (should_serialize<dm>(t.[:dm:])) {
+        if constexpr (simdjson::detail::has_annotation(dm, ^^simdjson::detail::flatten_tag)) {
+          static_assert(std::meta::is_class_type(simdjson::detail::flattened_type(dm)));
+          using flattened = std::remove_cvref_t<decltype(t.[:dm:])>;
+          static_assert(!concepts::container_but_not_string<flattened> && !concepts::string_view_keyed_map<flattened> &&
+                        !concepts::appendable_containers<flattened> && !concepts::optional_type<flattened> &&
+                        !concepts::smart_pointer<flattened> && !std::is_same_v<flattened, std::string> &&
+                        !std::is_same_v<flattened, std::string_view> && !require_custom_serialization<flattened>,
+                        "simdjson::flatten requires a member whose type is a structure serialized member by member");
+          atom_fields(w, t.[:dm:], first);
+        } else {
+          // Copy the key as whole 16-byte blocks from a zero-padded copy (one
+          // load and one store); ensure() reserves the padded length, and the
+          // unchecked writer has slack past its bound. Prior related work:
+          // jsonifier copies a power-of-two padded key and advances the cursor
+          // by the real length (serialize_impl.hpp, packed_blitter,
+          // https://github.com/nihilai-collective/Jsonifier).
+          constexpr const char* key_name = simdjson::get_json_key_name<dm>();
+          constexpr size_t first_key_len = constevalutil::consteval_to_quoted_escaped(key_name).size() + 1;
+          constexpr size_t rest_key_len = first_key_len + 1;
+          constexpr auto first_key = std::define_static_string(
+              constevalutil::consteval_to_quoted_escaped(key_name) + ":" +
+              std::string(padded_key_length(first_key_len) - first_key_len, '\0'));
+          constexpr auto rest_key = std::define_static_string(
+              std::string(",") + constevalutil::consteval_to_quoted_escaped(key_name) + ":" +
+              std::string(padded_key_length(rest_key_len) - rest_key_len, '\0'));
+          if (!w.ensure(padded_key_length(rest_key_len))) { return; }
+          if (first) {
+            std::memcpy(w.ptr + w.pos, first_key, padded_key_length(first_key_len));
+            w.pos += first_key_len;
+          } else {
+            std::memcpy(w.ptr + w.pos, rest_key, padded_key_length(rest_key_len));
+            w.pos += rest_key_len;
+          }
+          first = false;
+          atom_member<dm>(w, t.[:dm:]);
+        }
+      }
+    }
+  };
+}
+
+} // namespace annotation_detail
+
+template <class W, class T>
   requires(concepts::container_but_not_string<T> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &t) {
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
   auto it = t.begin();
   auto end = t.end();
   if (it == end) {
-    b.append_raw("[]");
+    if (!w.ensure(2)) return;
+    std::memcpy(w.ptr + w.pos, "[]", 2);
+    w.pos += 2;
     return;
   }
-  b.append('[');
-  atom(b, *it);
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = '[';
+  atom(w, *it);
   ++it;
   for (; it != end; ++it) {
-    b.append(',');
-    atom(b, *it);
+    if (!w.ensure(1)) return;
+    w.ptr[w.pos++] = ',';
+    atom(w, *it);
   }
-  b.append(']');
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = ']';
 }

-template <class T>
+template <class W, class T>
   requires(std::is_same_v<T, std::string> ||
            std::is_same_v<T, std::string_view> ||
            std::is_same_v<T, const char *> ||
            std::is_same_v<T, char>)
-constexpr void atom(string_builder &b, const T &t) {
-  b.escape_and_append_with_quotes(t);
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
+  // Inline the escape path through the writer so we never round-trip
+  // pos through memory for string fields (Twitter is dominated by
+  // these -- sync/reload around each string was a real cost).
+  std::string_view input;
+  if constexpr (std::is_same_v<T, char>) {
+    input = std::string_view(&t, 1);
+  } else {
+    input = std::string_view(t);
+  }
+  // Worst-case escape: every byte expands to \uXXXX (6 chars), plus 2 quotes.
+  // Guard against w.pos + 2 + 6 * input.size() wrapping for huge inputs -- if
+  // it wrapped to a small value, ensure() would spuriously succeed and the
+  // subsequent escape would overflow the buffer. max - w.pos cannot wrap, and
+  // size < (max - pos) / 6 implies pos + 6 * size + 6 <= max.
+  // Note that this is pedantic except maybe on 32-bit targets.
+  if constexpr (W::checked) {
+    if (simdjson_unlikely(input.size() >= ((std::numeric_limits<size_t>::max)() - w.pos) / 6)) { return; }
+    if (!w.ensure(2 + 6 * input.size())) { return; }
+  }
+  w.ptr[w.pos++] = '"';
+  w.pos += write_string_escaped(input, w.ptr + w.pos);
+  w.ptr[w.pos++] = '"';
 }

-template <concepts::string_view_keyed_map T>
+template <class W, concepts::string_view_keyed_map T>
   requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &m) {
+simdjson_really_inline constexpr void atom(W &w, const T &m) {
   if (m.empty()) {
-    b.append_raw("{}");
+    if (!w.ensure(2)) return;
+    std::memcpy(w.ptr + w.pos, "{}", 2);
+    w.pos += 2;
     return;
   }
-  b.append('{');
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = '{';
   bool first = true;
   for (const auto& [key, value] : m) {
     if (!first) {
-      b.append(',');
+      if (!w.ensure(1)) return;
+      w.ptr[w.pos++] = ',';
     }
     first = false;
-    // Keys must be convertible to string_view per the concept
-    b.escape_and_append_with_quotes(key);
-    b.append(':');
-    atom(b, value);
+    // Keys must be convertible to string_view per the concept.
+    std::string_view key_sv(key);
+    // Guard against w.pos + 3 + 6 * key_sv.size() wrapping for huge keys, if
+    // it wrapped to a small value, ensure() would spuriously succeed and the
+    // subsequent escape would overflow the buffer. max - w.pos cannot wrap, and
+    // size < (max - pos) / 6 implies pos + 6 * size + 6 <= max.
+    // Note that this is pedantic except maybe on 32-bit targets.
+    if constexpr (W::checked) {
+      if (simdjson_unlikely(key_sv.size() >= ((std::numeric_limits<size_t>::max)() - w.pos) / 6)) { return; }
+      if (!w.ensure(2 + 6 * key_sv.size() + 1)) { return; }
+    }
+    w.ptr[w.pos++] = '"';
+    w.pos += write_string_escaped(key_sv, w.ptr + w.pos);
+    w.ptr[w.pos++] = '"';
+    w.ptr[w.pos++] = ':';
+    atom(w, value);
   }
-  b.append('}');
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = '}';
 }


-template<typename number_type,
+template<class W, typename number_type,
          typename = typename std::enable_if<std::is_arithmetic<number_type>::value && !std::is_same_v<number_type, char>>::type>
-constexpr void atom(string_builder &b, const number_type t) {
-  b.append(t);
+simdjson_really_inline constexpr void atom(W &w, const number_type t) {
+  // Booleans / floats: defer to string_builder (rare path; keeps writer hot
+  // path free of float-formatter machinery). For integers, write directly
+  // via jeaiii using local pos.
+  if constexpr (std::is_same_v<number_type, bool>) {
+    if (t) {
+      if (!w.ensure(4)) return;
+      std::memcpy(w.ptr + w.pos, "true", 4);
+      w.pos += 4;
+    } else {
+      if (!w.ensure(5)) return;
+      std::memcpy(w.ptr + w.pos, "false", 5);
+      w.pos += 5;
+    }
+  } else if constexpr (std::is_floating_point_v<number_type>) {
+    if constexpr (W::checked) {
+      call_through_string_builder(w, [&](string_builder &b) { b.append(t); });
+    } else {
+      w.pos = size_t(internal::write_double(w.ptr + w.pos, double(t)) - w.ptr);
+    }
+  } else if constexpr (std::is_unsigned_v<number_type>) {
+    if (!w.ensure(20)) return;
+    char *end = internal::write_uint_jeaiii(
+        w.ptr + w.pos, static_cast<uint64_t>(t));
+    w.pos = static_cast<size_t>(end - w.ptr);
+  } else {
+    // signed integral
+    if (!w.ensure(20)) return;
+    using U = typename std::make_unsigned<number_type>::type;
+    bool negative = t < 0;
+    U pv = negative ? U(0) - static_cast<U>(t) : static_cast<U>(t);
+    w.ptr[w.pos] = '-';
+    w.pos += negative;
+    char *end = internal::write_uint_jeaiii(
+        w.ptr + w.pos, static_cast<uint64_t>(pv));
+    w.pos = static_cast<size_t>(end - w.ptr);
+  }
 }

-template <class T>
+template <class W, class T>
   requires(std::is_class_v<T> && !concepts::container_but_not_string<T> &&
            !concepts::string_view_keyed_map<T> &&
            !concepts::optional_type<T> &&
@@ -41554,92 +49970,259 @@ template <class T>
            !std::is_same_v<T, std::string_view> &&
            !std::is_same_v<T, const char*> &&
            !std::is_same_v<T, char> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &t) {
-  int i = 0;
-  b.append('{');
-  template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
-    if (i != 0)
-      b.append(',');
-    constexpr auto key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(dm)));
-    b.append_raw(key);
-    b.append(':');
-    atom(b, t.[:dm:]);
-    i++;
-  };
-  b.append('}');
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
+  if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag)) {
+    // A transparent structure is serialized as its single member.
+    constexpr auto dm = simdjson::detail::transparent_member(^^T);
+    annotation_detail::atom_member<dm>(w, t.[:dm:]);
+  } else {
+    bool first = true;
+    if (!w.ensure(1)) { return; }
+    w.ptr[w.pos++] = '{';
+    annotation_detail::atom_fields(w, t, first);
+    if (!w.ensure(1)) { return; }
+    w.ptr[w.pos++] = '}';
+  }
 }

 // Support for optional types (std::optional, etc.)
-template <concepts::optional_type T>
+template <class W, concepts::optional_type T>
   requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &opt) {
+simdjson_really_inline constexpr void atom(W &w, const T &opt) {
   if (opt) {
-    atom(b, opt.value());
+    atom(w, opt.value());
   } else {
-    b.append_raw("null");
+    if (!w.ensure(4)) return;
+    std::memcpy(w.ptr + w.pos, "null", 4);
+    w.pos += 4;
   }
 }

 // Support for smart pointers (std::unique_ptr, std::shared_ptr, etc.)
-template <concepts::smart_pointer T>
+template <class W, concepts::smart_pointer T>
   requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &ptr) {
+simdjson_really_inline constexpr void atom(W &w, const T &ptr) {
   if (ptr) {
-    atom(b, *ptr);
+    atom(w, *ptr);
   } else {
-    b.append_raw("null");
+    if (!w.ensure(4)) return;
+    std::memcpy(w.ptr + w.pos, "null", 4);
+    w.pos += 4;
   }
 }

 // Support for enums - serialize as string representation using expand approach from P2996R12
-template <typename T>
+template <class W, typename T>
   requires(std::is_enum_v<T> && !require_custom_serialization<T>)
-void atom(string_builder &b, const T &e) {
+simdjson_really_inline void atom(W &w, const T &e) {
 #if SIMDJSON_STATIC_REFLECTION
-  constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
+  static constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
   template for (constexpr auto enum_val : enumerators) {
-    constexpr auto enum_str = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(enum_val)));
+    constexpr auto enum_str = std::define_static_string(constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<enum_val>()));
+    constexpr size_t enum_str_len = std::char_traits<char>::length(enum_str);
     if (e == [:enum_val:]) {
-      b.append_raw(enum_str);
+      if (!w.ensure(enum_str_len)) return;
+      std::memcpy(w.ptr + w.pos, enum_str, enum_str_len);
+      w.pos += enum_str_len;
       return;
     }
   };
   // Fallback to integer if enum value not found
-  atom(b, static_cast<std::underlying_type_t<T>>(e));
+  atom(w, static_cast<std::underlying_type_t<T>>(e));
 #else
   // Fallback: serialize as integer if reflection not available
-  atom(b, static_cast<std::underlying_type_t<T>>(e));
+  atom(w, static_cast<std::underlying_type_t<T>>(e));
 #endif
 }

 // Support for appendable containers that don't have operator[] (sets, etc.)
-template <concepts::appendable_containers T>
+template <class W, concepts::appendable_containers T>
   requires(!concepts::container_but_not_string<T> && !concepts::string_view_keyed_map<T> &&
            !concepts::optional_type<T> && !concepts::smart_pointer<T> &&
            !std::is_same_v<T, std::string> &&
            !std::is_same_v<T, std::string_view> && !std::is_same_v<T, const char*> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &container) {
+simdjson_really_inline constexpr void atom(W &w, const T &container) {
   if (container.empty()) {
-    b.append_raw("[]");
+    if (!w.ensure(2)) return;
+    std::memcpy(w.ptr + w.pos, "[]", 2);
+    w.pos += 2;
     return;
   }
-  b.append('[');
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = '[';
   bool first = true;
   for (const auto& item : container) {
     if (!first) {
-      b.append(',');
+      if (!w.ensure(1)) return;
+      w.ptr[w.pos++] = ',';
     }
     first = false;
-    atom(b, item);
+    atom(w, item);
+  }
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = ']';
+}
+
+// =============================================================
+// Size bound: an upper bound on the number of bytes that atom(w, t) writes.
+// Computing it first lets append() reserve the capacity once and then run
+// the whole write chain through an unchecked_writer, without a capacity
+// check before every write. It mirrors the atom() overloads above.
+//
+// Prior related work: jsonifier sizes the document first, counting 6 bytes
+// per string byte, resizes once to that bound plus slack, and writes with
+// no capacity check on each store. to_json() does that resize through
+// std::string::resize_and_overwrite. See serializer.hpp and
+// serialize_impl.hpp in https://github.com/nihilai-collective/Jsonifier
+// and https://nihilai-collective.net/serialization.
+// =============================================================
+namespace bound_detail {
+
+// Whether size_bound covers everything that atom() writes for T: not when a
+// member is serialized by a with<Adapter> serializer, which writes an unknown
+// amount through the string_builder.
+template <class T>
+consteval bool is_bounded() {
+  if constexpr (require_custom_serialization<T>) {
+    return false;
+  } else if constexpr (std::is_same_v<T, std::string> || std::is_same_v<T, std::string_view> ||
+                       std::is_same_v<T, const char *> || std::is_arithmetic_v<T> || std::is_enum_v<T>) {
+    return true;
+  } else if constexpr (concepts::optional_type<T> || concepts::smart_pointer<T>) {
+    return is_bounded<std::remove_cvref_t<decltype(*std::declval<const T &>())>>();
+  } else if constexpr (concepts::string_view_keyed_map<T>) {
+    return is_bounded<std::remove_cvref_t<typename T::mapped_type>>();
+  } else if constexpr (concepts::container_but_not_string<T> || concepts::appendable_containers<T>) {
+    return is_bounded<std::remove_cvref_t<std::ranges::range_value_t<T>>>();
+  } else {
+    bool bounded = true;
+    template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+      if constexpr (annotation_detail::is_serialized_member(dm)) {
+        bounded = bounded && simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::with_t) == std::meta::info{} &&
+                  is_bounded<std::remove_cvref_t<decltype(std::declval<const T &>().[:dm:])>>();
+      }
+    };
+    return bounded;
+  }
+}
+
+template <class T>
+consteval size_t enum_bound() {
+  size_t bound = 20; // the integer fallback
+  template for (constexpr auto enum_val : std::define_static_array(std::meta::enumerators_of(^^T))) {
+    constexpr size_t len = std::char_traits<char>::length(std::define_static_string(
+        constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<enum_val>())));
+    bound = (std::max)(bound, len);
+  };
+  return bound;
+}
+
+template <class T>
+simdjson_really_inline size_t size_bound(const T &t) noexcept;
+
+// Bound for the "key":value pairs of a structure, commas included.
+template <class T>
+simdjson_really_inline size_t fields_bound(const T &t) noexcept {
+  size_t bound = 0;
+  template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+    if constexpr (annotation_detail::is_serialized_member(dm)) {
+      if constexpr (simdjson::detail::has_annotation(dm, ^^simdjson::detail::flatten_tag)) {
+        bound += fields_bound(t.[:dm:]);
+      } else {
+        constexpr size_t rest_key_len = constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<dm>()).size() + 2;
+        bound += rest_key_len + size_bound(t.[:dm:]);
+      }
+    }
+  };
+  return bound;
+}
+
+template <class T>
+simdjson_really_inline size_t size_bound([[maybe_unused]] const T &t) noexcept {
+  if constexpr (std::is_same_v<T, char>) {
+    return 2 + 6;
+  } else if constexpr (std::is_same_v<T, std::string> || std::is_same_v<T, std::string_view> ||
+                       std::is_same_v<T, const char *>) {
+    // Every byte may become \uXXXX, plus the quotes.
+    return 2 + 6 * std::string_view(t).size();
+  } else if constexpr (std::is_same_v<T, bool>) {
+    return 5;
+  } else if constexpr (std::is_floating_point_v<T>) {
+    return simdjson::internal::to_chars_buffer_size;
+  } else if constexpr (std::is_arithmetic_v<T>) {
+    return 20;
+  } else if constexpr (std::is_enum_v<T>) {
+    return enum_bound<T>();
+  } else if constexpr (concepts::optional_type<T> || concepts::smart_pointer<T>) {
+    return t ? size_bound(*t) : 4;
+  } else if constexpr (concepts::string_view_keyed_map<T>) {
+    size_t bound = 2;
+    for (const auto &[key, value] : t) {
+      // comma, quotes, colon
+      bound += 4 + 6 * std::string_view(key).size() + size_bound(value);
+    }
+    return bound;
+  } else if constexpr (concepts::container_but_not_string<T> || concepts::appendable_containers<T>) {
+    using value_type = std::remove_cvref_t<std::ranges::range_value_t<T>>;
+    if constexpr (std::is_arithmetic_v<value_type> && !std::is_same_v<value_type, char>) {
+      // A fixed bound per element: no need to visit them.
+      return 2 + size_t(std::ranges::distance(t)) * (1 + size_bound(value_type{}));
+    } else {
+      size_t bound = 2;
+#if defined(__GNUC__) && !defined(__clang__)
+#pragma GCC novector // a vector loop is slower on short containers
+#endif
+#pragma GCC unroll 4
+      for (const auto &item : t) {
+        bound += 1 + size_bound(item);
+      }
+      return bound;
+    }
+  } else if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag)) {
+    constexpr auto dm = simdjson::detail::transparent_member(^^T);
+    return size_bound(t.[:dm:]);
+  } else {
+    return 2 + fields_bound(t);
+  }
+}
+
+} // namespace bound_detail
+
+// Write t through an unchecked writer when its size bound is available,
+// reserving that many bytes first, and through the checked writer otherwise.
+template <class T>
+simdjson_really_inline void append_bounded(string_builder &b, const T &t) {
+  // On 32-bit systems, the bound could overflow: keep the checked writer.
+  if constexpr (sizeof(size_t) >= 8 && bound_detail::is_bounded<T>()) {
+    const size_t bound = bound_detail::size_bound(t) + unchecked_slack;
+    const size_t pos = b.unsafe_position();
+    // The bound is a sum of in-memory sizes times a small constant: it cannot
+    // overflow on a 64-bit system. Be pedantic elsewhere.
+    if (sizeof(size_t) >= 8 || bound <= (std::numeric_limits<size_t>::max)() - pos) {
+      const size_t cap = b.unsafe_capacity();
+      // Grow geometrically so that many small appends stay amortized.
+      if (pos + bound <= cap || b.unsafe_grow((std::max)(cap * 2, pos + bound))) {
+        unchecked_writer w(b.unsafe_data(), pos);
+        atom(w, t);
+        b.unsafe_set_position(w.pos);
+      }
+      return;
+    }
   }
-  b.append(']');
+  writer w(b);
+  atom(w, t);
+  w.sync();
 }

-// append functions that delegate to atom functions for primitive types
+// append() -- top-level entry. Each overload constructs a stack-local
+// writer, runs atom(w, t) through the inlined call chain, then syncs
+// the local position back into the string_builder.
 template <class T>
   requires(std::is_arithmetic_v<T> && !std::is_same_v<T, char>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  writer w(b);
+  atom(w, t);
+  w.sync();
 }

 template <class T>
@@ -41647,20 +50230,22 @@ template <class T>
            std::is_same_v<T, std::string_view> ||
            std::is_same_v<T, const char *> ||
            std::is_same_v<T, char>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  writer w(b);
+  atom(w, t);
+  w.sync();
 }

 template <concepts::optional_type T>
   requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  append_bounded(b, t);
 }

 template <concepts::smart_pointer T>
   requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  append_bounded(b, t);
 }

 template <concepts::appendable_containers T>
@@ -41668,14 +50253,14 @@ template <concepts::appendable_containers T>
            !concepts::optional_type<T> && !concepts::smart_pointer<T> &&
            !std::is_same_v<T, std::string> &&
            !std::is_same_v<T, std::string_view> && !std::is_same_v<T, const char*> && !require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  append_bounded(b, t);
 }

 template <concepts::string_view_keyed_map T>
   requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  append_bounded(b, t);
 }

 // works for struct
@@ -41689,39 +50274,15 @@ template <class Z>
            !std::is_same_v<Z, std::string_view> &&
            !std::is_same_v<Z, const char*> &&
            !std::is_same_v<Z, char> && !require_custom_serialization<Z>)
-void append(string_builder &b, const Z &z) {
-  int i = 0;
-  b.append('{');
-  template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^Z, std::meta::access_context::unchecked()))) {
-    if (i != 0)
-      b.append(',');
-    constexpr auto key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(dm)));
-    b.append_raw(key);
-    b.append(':');
-    atom(b, z.[:dm:]);
-    i++;
-  };
-  b.append('}');
+simdjson_inline void append(string_builder &b, const Z &z) {
+  append_bounded(b, z);
 }

 // works for container that have begin() and end() iterators
 template <class Z>
   requires(concepts::container_but_not_string<Z> && !require_custom_serialization<Z>)
-void append(string_builder &b, const Z &z) {
-  auto it = z.begin();
-  auto end = z.end();
-  if (it == end) {
-    b.append_raw("[]");
-    return;
-  }
-  b.append('[');
-  atom(b, *it);
-  ++it;
-  for (; it != end; ++it) {
-    b.append(',');
-    atom(b, *it);
-  }
-  b.append(']');
+simdjson_inline void append(string_builder &b, const Z &z) {
+  append_bounded(b, z);
 }

 template <class Z>
@@ -41732,22 +50293,40 @@ void append(string_builder &b, const Z &z) {


 template <class Z>
-simdjson_warn_unused simdjson_result<std::string> to_json_string(const Z &z, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
-  string_builder b(initial_capacity);
-  append(b, z);
-  std::string_view s;
-  if(auto e = b.view().get(s); e) { return e; }
-  return std::string(s);
+simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
+  if constexpr (sizeof(size_t) >= 8 && bound_detail::is_bounded<Z>()) {
+    // Write straight into s, sized by the bound: no intermediate buffer, no copy.
+    // Prior related work: jsonifier's serializeJson resizes once through
+    // resize_and_overwrite (serializer.hpp).
+    (void)initial_capacity;
+    const size_t bound = bound_detail::size_bound(z) + unchecked_slack;
+    auto write = [&z](char *p) noexcept {
+      unchecked_writer w(p, 0);
+      atom(w, z);
+      return w.pos;
+    };
+#if defined(__cpp_lib_string_resize_and_overwrite) && __cpp_lib_string_resize_and_overwrite >= 202110L
+    s.resize_and_overwrite(bound, [&write](char *p, size_t) noexcept { return write(p); });
+#else
+    s.resize(bound);
+    s.resize(write(s.data()));
+#endif
+    return SUCCESS;
+  } else {
+    string_builder b(initial_capacity);
+    append(b, z);
+    std::string_view view;
+    if(auto e = b.view().get(view); e) { return e; }
+    s.assign(view);
+    return SUCCESS;
+  }
 }

 template <class Z>
-simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
-  string_builder b(initial_capacity);
-  append(b, z);
-  std::string_view view;
-  if(auto e = b.view().get(view); e) { return e; }
-  s.assign(view);
-  return SUCCESS;
+simdjson_warn_unused simdjson_result<std::string> to_json_string(const Z &z, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
+  std::string s;
+  if(auto e = to_json(z, s, initial_capacity); e) { return e; }
+  return s;
 }

 template <class Z>
@@ -41760,40 +50339,41 @@ string_builder& operator<<(string_builder& b, const Z& z) {
 template<constevalutil::fixed_string... FieldNames, typename T>
   requires(std::is_class_v<T> && (sizeof...(FieldNames) > 0))
 void extract_from(string_builder &b, const T &obj) {
-  // Helper to check if a field name matches any of the requested fields
-  auto should_extract = [](std::string_view field_name) constexpr -> bool {
-    return ((FieldNames.view() == field_name) || ...);
-  };
-
-  b.append('{');
+  writer w(b);
+  if (!w.ensure(1)) { w.sync(); return; }
+  w.ptr[w.pos++] = '{';
   bool first = true;
-
   // Iterate through all members of T using reflection
-  template for (constexpr auto mem : std::define_static_array(
-      std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
-
+  static constexpr auto members = std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()));
+  template for (constexpr auto mem : members) {
     if constexpr (std::meta::is_public(mem)) {
-      constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
+      static constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));

       // Only serialize this field if it's in our list of requested fields
-      if constexpr (should_extract(key)) {
-        if (!first) {
-          b.append(',');
+      if constexpr (((FieldNames.view() == key) || ...)) {
+        static constexpr auto first_key = std::define_static_string(
+            constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)) + ":");
+        static constexpr auto rest_key = std::define_static_string(
+            std::string(",") + constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)) + ":");
+        constexpr size_t first_key_len = std::char_traits<char>::length(first_key);
+        constexpr size_t rest_key_len = std::char_traits<char>::length(rest_key);
+        if (!w.ensure(rest_key_len)) { w.sync(); return; }
+        if (first) {
+          std::memcpy(w.ptr + w.pos, first_key, first_key_len);
+          w.pos += first_key_len;
+        } else {
+          std::memcpy(w.ptr + w.pos, rest_key, rest_key_len);
+          w.pos += rest_key_len;
         }
         first = false;
-
-        // Serialize the key
-        constexpr auto quoted_key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)));
-        b.append_raw(quoted_key);
-        b.append(':');
-
-        // Serialize the value
-        atom(b, obj.[:mem:]);
+        atom(w, obj.[:mem:]);
       }
     }
   };

-  b.append('}');
+  if (!w.ensure(1)) { w.sync(); return; }
+  w.ptr[w.pos++] = '}';
+  w.sync();
 }

 template<constevalutil::fixed_string... FieldNames, typename T>
@@ -41806,25 +50386,18 @@ simdjson_warn_unused simdjson_result<std::string> extract_from(const T &obj, siz
   return std::string(s);
 }

+SIMDJSON_POP_DISABLE_WARNINGS
+
 } // namespace builder
 } // namespace fallback
 // Alias the function template to 'to' in the global namespace
 template <class Z>
 simdjson_warn_unused simdjson_result<std::string> to_json(const Z &z, size_t initial_capacity = fallback::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
-  fallback::builder::string_builder b(initial_capacity);
-  fallback::builder::append(b, z);
-  std::string_view s;
-  if(auto e = b.view().get(s); e) { return e; }
-  return std::string(s);
+  return fallback::builder::to_json_string(z, initial_capacity);
 }
 template <class Z>
 simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = fallback::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
-  fallback::builder::string_builder b(initial_capacity);
-  fallback::builder::append(b, z);
-  std::string_view view;
-  if(auto e = b.view().get(view); e) { return e; }
-  s.assign(view);
-  return SUCCESS;
+  return fallback::builder::to_json(z, s, initial_capacity);
 }
 // Global namespace function for extract_from
 template<constevalutil::fixed_string... FieldNames, typename T>
@@ -41970,6 +50543,7 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
 /* including simdjson/generic/builder/json_string_builder-inl.h for fallback: #include "simdjson/generic/builder/json_string_builder-inl.h" */
 /* begin file simdjson/generic/builder/json_string_builder-inl.h for fallback */
 #include <array>
+#include <cmath>
 #include <cstring>
 #include <limits>
 #include <type_traits>
@@ -42004,6 +50578,11 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
 #define SIMDJSON_EXPERIMENTAL_HAS_LSX 1
 #endif
 #endif
+#if defined(__loongarch_asx)
+#ifndef SIMDJSON_EXPERIMENTAL_HAS_LASX
+#define SIMDJSON_EXPERIMENTAL_HAS_LASX 1
+#endif
+#endif
 #if defined(__riscv_v_intrinsic) && __riscv_v_intrinsic >= 11000 &&            \
     defined(__riscv_vector)
 #ifndef SIMDJSON_EXPERIMENTAL_HAS_RVV
@@ -42023,6 +50602,9 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
 #endif
 #if SIMDJSON_EXPERIMENTAL_HAS_SSE2
 #include <emmintrin.h>
+#if defined(__AVX2__)
+#include <immintrin.h>
+#endif
 #ifdef _MSC_VER
 #include <intrin.h>
 #endif
@@ -42030,6 +50612,9 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
 #if SIMDJSON_EXPERIMENTAL_HAS_LSX
 #include <lsxintrin.h>
 #endif
+#if SIMDJSON_EXPERIMENTAL_HAS_LASX
+#include <lasxintrin.h>
+#endif
 #if SIMDJSON_EXPERIMENTAL_HAS_RVV
 #include <riscv_vector.h>
 #endif
@@ -42079,105 +50664,6 @@ inline bool has_json_escapable_byte(uint64_t x) {

 **/

-SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline bool
-simple_needs_escaping(std::string_view v) {
-  for (char c : v) {
-    // a table lookup is faster than a series of comparisons
-    if (json_quotable_character[static_cast<uint8_t>(c)]) {
-      return true;
-    }
-  }
-  return false;
-}
-
-#if SIMDJSON_EXPERIMENTAL_HAS_NEON
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
-  if (view.size() < 16) {
-    return simple_needs_escaping(view);
-  }
-  size_t i = 0;
-  uint8x16_t running = vdupq_n_u8(0);
-  uint8x16_t v34 = vdupq_n_u8(34);
-  uint8x16_t v92 = vdupq_n_u8(92);
-
-  for (; i + 15 < view.size(); i += 16) {
-    uint8x16_t word = vld1q_u8((const uint8_t *)view.data() + i);
-    running = vorrq_u8(running, vceqq_u8(word, v34));
-    running = vorrq_u8(running, vceqq_u8(word, v92));
-    running = vorrq_u8(running, vcltq_u8(word, vdupq_n_u8(32)));
-  }
-  if (i < view.size()) {
-    uint8x16_t word =
-        vld1q_u8((const uint8_t *)view.data() + view.length() - 16);
-    running = vorrq_u8(running, vceqq_u8(word, v34));
-    running = vorrq_u8(running, vceqq_u8(word, v92));
-    running = vorrq_u8(running, vcltq_u8(word, vdupq_n_u8(32)));
-  }
-  return vmaxvq_u32(vreinterpretq_u32_u8(running)) != 0;
-}
-#elif SIMDJSON_EXPERIMENTAL_HAS_SSE2
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
-  if (view.size() < 16) {
-    return simple_needs_escaping(view);
-  }
-  size_t i = 0;
-  __m128i running = _mm_setzero_si128();
-  for (; i + 15 < view.size(); i += 16) {
-
-    __m128i word =
-        _mm_loadu_si128(reinterpret_cast<const __m128i *>(view.data() + i));
-    running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(34)));
-    running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(92)));
-    running = _mm_or_si128(
-        running, _mm_cmpeq_epi8(_mm_subs_epu8(word, _mm_set1_epi8(31)),
-                                _mm_setzero_si128()));
-  }
-  if (i < view.size()) {
-    __m128i word = _mm_loadu_si128(
-        reinterpret_cast<const __m128i *>(view.data() + view.length() - 16));
-    running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(34)));
-    running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(92)));
-    running = _mm_or_si128(
-        running, _mm_cmpeq_epi8(_mm_subs_epu8(word, _mm_set1_epi8(31)),
-                                _mm_setzero_si128()));
-  }
-  return _mm_movemask_epi8(running) != 0;
-}
-#elif SIMDJSON_EXPERIMENTAL_HAS_PPC64
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
-  if (view.size() < 16) {
-    return simple_needs_escaping(view);
-  }
-  size_t i = 0;
-  __vector unsigned char running = vec_splats((unsigned char)0);
-  __vector unsigned char v34 = vec_splats((unsigned char)34);
-  __vector unsigned char v92 = vec_splats((unsigned char)92);
-  __vector unsigned char v32 = vec_splats((unsigned char)32);
-
-  for (; i + 15 < view.size(); i += 16) {
-    __vector unsigned char word =
-        vec_vsx_ld(0, reinterpret_cast<const unsigned char *>(view.data() + i));
-    running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v34));
-    running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v92));
-    running = vec_or(running,
-        (__vector unsigned char)vec_cmplt(word, v32));
-  }
-  if (i < view.size()) {
-    __vector unsigned char word = vec_vsx_ld(
-        0, reinterpret_cast<const unsigned char *>(view.data() + view.length() - 16));
-    running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v34));
-    running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v92));
-    running = vec_or(running,
-        (__vector unsigned char)vec_cmplt(word, v32));
-  }
-  return !vec_all_eq(running, vec_splats((unsigned char)0));
-}
-#else
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
-  return simple_needs_escaping(view);
-}
-#endif
-
 // Scalar fallback for finding next quotable character
 SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline size_t
 find_next_json_quotable_character_scalar(const std::string_view view,
@@ -42268,6 +50754,51 @@ find_next_json_quotable_character(const std::string_view view,
   size_t current = len - remaining;
   return find_next_json_quotable_character_scalar(view, current);
 }
+#elif SIMDJSON_EXPERIMENTAL_HAS_LASX
+simdjson_inline size_t
+find_next_json_quotable_character(const std::string_view view,
+                                  size_t location) noexcept {
+  const size_t len = view.size();
+  const uint8_t *ptr =
+      reinterpret_cast<const uint8_t *>(view.data()) + location;
+  size_t remaining = len - location;
+
+  // SIMD constants for characters requiring escape
+  __m256i v34 = __lasx_xvreplgr2vr_b(34);  // '"'
+  __m256i v92 = __lasx_xvreplgr2vr_b(92);  // '\\'
+  __m256i v32 = __lasx_xvreplgr2vr_b(32);  // control char threshold
+
+  while (remaining >= 32) {
+    __m256i word = __lasx_xvld(ptr, 0);
+
+    // Check for quotable characters: '"', '\\', or control chars (< 32)
+    __m256i needs_escape = __lasx_xvseq_b(word, v34);
+    needs_escape = __lasx_xvor_v(needs_escape, __lasx_xvseq_b(word, v92));
+    needs_escape = __lasx_xvor_v(needs_escape, __lasx_xvslt_bu(word, v32));
+
+    if (!__lasx_xbz_v(needs_escape)) {
+      // Found a quotable character - locate it via the four 64-bit lanes
+      uint64_t lane0 = __lasx_xvpickve2gr_du(needs_escape, 0);
+      uint64_t lane1 = __lasx_xvpickve2gr_du(needs_escape, 1);
+      uint64_t lane2 = __lasx_xvpickve2gr_du(needs_escape, 2);
+      uint64_t lane3 = __lasx_xvpickve2gr_du(needs_escape, 3);
+      size_t offset = ptr - reinterpret_cast<const uint8_t *>(view.data());
+      if (lane0 != 0) {
+        return offset + trailing_zeroes(lane0) / 8;
+      } else if (lane1 != 0) {
+        return offset + 8 + trailing_zeroes(lane1) / 8;
+      } else if (lane2 != 0) {
+        return offset + 16 + trailing_zeroes(lane2) / 8;
+      } else {
+        return offset + 24 + trailing_zeroes(lane3) / 8;
+      }
+    }
+    ptr += 32;
+    remaining -= 32;
+  }
+  size_t current = len - remaining;
+  return find_next_json_quotable_character_scalar(view, current);
+}
 #elif SIMDJSON_EXPERIMENTAL_HAS_LSX
 simdjson_inline size_t
 find_next_json_quotable_character(const std::string_view view,
@@ -42423,6 +50954,254 @@ SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline void escape_json_char(char c, char *&o
   }
 }

+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2 || SIMDJSON_EXPERIMENTAL_HAS_NEON
+#define SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE 1
+#endif
+
+#if SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2
+
+using escape_vector = __m128i;
+// Mask bits that each input byte contributes to escape_bitmask().
+static constexpr unsigned escape_mask_bits = 1;
+
+simdjson_inline escape_vector escape_load16(const uint8_t *p) noexcept {
+  return _mm_loadu_si128(reinterpret_cast<const __m128i *>(p));
+}
+
+simdjson_inline void escape_store16(char *out, escape_vector v) noexcept {
+  _mm_storeu_si128(reinterpret_cast<__m128i *>(out), v);
+}
+
+// Builds a vector whose bytes 0..7 come from a and bytes 8..15 from b.
+simdjson_inline escape_vector escape_load8x2(const uint8_t *a,
+                                             const uint8_t *b) noexcept {
+  return _mm_unpacklo_epi64(
+      _mm_loadl_epi64(reinterpret_cast<const __m128i *>(a)),
+      _mm_loadl_epi64(reinterpret_cast<const __m128i *>(b)));
+}
+
+// Builds a vector whose bytes 0..3 come from a and bytes 4..7 from b. The
+// upper half is unspecified; callers only look at the low eight bits.
+simdjson_inline escape_vector escape_load4x2(const uint8_t *a,
+                                             const uint8_t *b) noexcept {
+  int32_t a32, b32;
+  memcpy(&a32, a, 4);
+  memcpy(&b32, b, 4);
+  return _mm_unpacklo_epi32(_mm_cvtsi32_si128(a32), _mm_cvtsi32_si128(b32));
+}
+
+// Sets every bit of byte k when byte k of v is a quotable character ('"',
+// '\\' or a control character).
+simdjson_inline escape_vector escape_flags(escape_vector v) noexcept {
+  const __m128i v34 = _mm_set1_epi8(34); // '"'
+  const __m128i v92 = _mm_set1_epi8(92); // '\\'
+  const __m128i v31 = _mm_set1_epi8(31); // for control char detection
+  __m128i needs_escape = _mm_cmpeq_epi8(v, v34);
+  needs_escape = _mm_or_si128(needs_escape, _mm_cmpeq_epi8(v, v92));
+  return _mm_or_si128(
+      needs_escape, _mm_cmpeq_epi8(_mm_subs_epu8(v, v31), _mm_setzero_si128()));
+}
+
+// True when any byte needs escaping. Kept separate from escape_bitmask
+// because some instruction sets can answer it without leaving the vector
+// register file; on SSE2 the compiler folds the two together.
+simdjson_inline bool escape_any(escape_vector flags) noexcept {
+  return _mm_movemask_epi8(flags) != 0;
+}
+
+// A 16-bit mask with bit k set when byte k needed escaping.
+simdjson_inline uint64_t escape_bitmask(escape_vector flags) noexcept {
+  return uint64_t(uint32_t(_mm_movemask_epi8(flags)));
+}
+
+#else // SIMDJSON_EXPERIMENTAL_HAS_NEON
+
+using escape_vector = uint8x16_t;
+static constexpr unsigned escape_mask_bits = 4;
+
+simdjson_inline escape_vector escape_load16(const uint8_t *p) noexcept {
+  return vld1q_u8(p);
+}
+
+simdjson_inline void escape_store16(char *out, escape_vector v) noexcept {
+  vst1q_u8(reinterpret_cast<uint8_t *>(out), v);
+}
+
+simdjson_inline escape_vector escape_load8x2(const uint8_t *a,
+                                             const uint8_t *b) noexcept {
+  uint64_t a64, b64;
+  memcpy(&a64, a, 8);
+  memcpy(&b64, b, 8);
+  return vreinterpretq_u8_u64(vsetq_lane_u64(b64, vdupq_n_u64(a64), 1));
+}
+
+simdjson_inline escape_vector escape_load4x2(const uint8_t *a,
+                                             const uint8_t *b) noexcept {
+  uint32_t a32, b32;
+  memcpy(&a32, a, 4);
+  memcpy(&b32, b, 4);
+  return vreinterpretq_u8_u32(vsetq_lane_u32(b32, vdupq_n_u32(a32), 1));
+}
+
+simdjson_inline escape_vector escape_flags(escape_vector v) noexcept {
+  uint8x16_t needs_escape = vceqq_u8(v, vdupq_n_u8(34));              // '"'
+  needs_escape = vorrq_u8(needs_escape, vceqq_u8(v, vdupq_n_u8(92))); // '\\'
+  return vorrq_u8(needs_escape, vcltq_u8(v, vdupq_n_u8(32)));
+}
+
+simdjson_inline bool escape_any(escape_vector flags) noexcept {
+  return vmaxvq_u32(vreinterpretq_u32_u8(flags)) != 0;
+}
+
+// Four bits per byte rather than one: escape_block and the tail paths scale
+// their shifts by escape_mask_bits to match.
+simdjson_inline uint64_t escape_bitmask(escape_vector flags) noexcept {
+  uint8x8_t narrowed = vshrn_n_u16(vreinterpretq_u16_u8(flags), 4);
+  return vget_lane_u64(vreinterpret_u64_u8(narrowed), 0);
+}
+
+#endif // instruction set selection
+
+// The escape bitmask of a 16-byte block.
+simdjson_inline uint64_t escape_mask(escape_vector v) noexcept {
+  return escape_bitmask(escape_flags(v));
+}
+
+// Copies n bytes with n < 16, using overlapping loads and stores. It never
+// reads more than n bytes from src, nor writes more than n bytes to dst.
+simdjson_inline void copy_lt16(char *dst, const uint8_t *src,
+                               size_t n) noexcept {
+  if (n >= 8) {
+    memcpy(dst, src, 8);
+    memcpy(dst + n - 8, src + n - 8, 8);
+  } else if (n >= 4) {
+    memcpy(dst, src, 4);
+    memcpy(dst + n - 4, src + n - 4, 4);
+  } else if (n > 0) {
+    dst[0] = char(src[0]);
+    dst[n >> 1] = char(src[n >> 1]);
+    dst[n - 1] = char(src[n - 1]);
+  }
+}
+
+// Escapes the bytes of src in the range [i, blockend), given that m is the
+// (non-zero) escape mask for that range: the escape_mask_bits-wide lane of m
+// at byte k is non-zero when src[i + k] requires escaping. Returns the updated
+// output pointer.
+simdjson_never_inline char *escape_block(const uint8_t *src, char *out,
+                                         size_t i, size_t blockend,
+                                         uint64_t m) noexcept {
+  constexpr uint64_t lane = (uint64_t(1) << escape_mask_bits) - 1;
+
+  size_t pos = i; // first byte not yet copied
+  while (m) {
+    const size_t tz = trailing_zeroes(m);
+    const size_t next = i + tz / escape_mask_bits;
+    // Copy the run of safe bytes that precedes this escape.
+    copy_lt16(out, src + pos, next - pos);
+    out += next - pos;
+    escape_json_char(char(src[next]), out);
+    pos = next + 1;
+    m &= ~(lane << tz);
+  }
+  // Copy whatever follows the last escape.
+  copy_lt16(out, src + pos, blockend - pos);
+  return out + (blockend - pos);
+}
+
+// Writes the escaped version of input to out, returning the number of bytes
+// written.
+simdjson_really_inline size_t write_string_escaped(const std::string_view input, char *out) {
+  const size_t len = input.size();
+  const uint8_t *src = reinterpret_cast<const uint8_t *>(input.data());
+  const char *const initout = out;
+
+  size_t i = 0;
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2 && defined(__AVX2__)
+  while (i + 32 <= len) {
+    const __m256i word = _mm256_loadu_si256(reinterpret_cast<const __m256i *>(src + i));
+    const __m256i flags = _mm256_or_si256(
+        _mm256_or_si256(_mm256_cmpeq_epi8(word, _mm256_set1_epi8(34)),   // '"'
+                        _mm256_cmpeq_epi8(word, _mm256_set1_epi8(92))),  // '\\'
+        _mm256_cmpeq_epi8(_mm256_subs_epu8(word, _mm256_set1_epi8(31)),
+                          _mm256_setzero_si256()));                      // control
+    const uint32_t mask = uint32_t(_mm256_movemask_epi8(flags));
+    if (simdjson_likely(mask == 0)) {
+      _mm256_storeu_si256(reinterpret_cast<__m256i *>(out), word);
+      out += 32;
+    } else {
+      for (size_t half = 0; half < 32; half += 16) {
+        const uint64_t m = (mask >> half) & 0xFFFF;
+        if (m == 0) {
+          escape_store16(out, escape_load16(src + i + half));
+          out += 16;
+        } else {
+          out = escape_block(src, out, i + half, i + half + 16, m);
+        }
+      }
+    }
+    i += 32;
+  }
+#endif
+  while (i + 16 <= len) {
+    escape_vector word = escape_load16(src + i);
+    escape_vector flags = escape_flags(word);
+    if (simdjson_likely(!escape_any(flags))) {
+      escape_store16(out, word);
+      out += 16;
+    } else {
+      out = escape_block(src, out, i, i + 16, escape_bitmask(flags));
+    }
+    i += 16;
+  }
+  if (i < len) {
+    const size_t rem = len - i;
+    uint64_t m;
+    if (len >= 16) {
+      // The last 16 bytes of the input are in bounds. Bit k of that block's
+      // mask belongs to input position len - 16 + k, so shift it down to align
+      // bit 0 with position i.
+      m = escape_mask(escape_load16(src + len - 16)) >>
+          (escape_mask_bits * (16 - rem));
+    } else if (len >= 8) {
+      // Here i == 0 and rem == len. Two overlapping 8-byte loads cover
+      // [0, 8) and [len - 8, len), which is the whole input since len < 16.
+      uint64_t mm = escape_mask(escape_load8x2(src, src + len - 8));
+      constexpr uint64_t low8 = (uint64_t(1) << (escape_mask_bits * 8)) - 1;
+      m = (mm & low8) |
+          ((mm >> (escape_mask_bits * 8)) << (escape_mask_bits * (len - 8)));
+    } else if (len >= 4) {
+      // Same idea with two overlapping 4-byte loads.
+      uint64_t mm = escape_mask(escape_load4x2(src, src + len - 4));
+      constexpr uint64_t low4 = (uint64_t(1) << (escape_mask_bits * 4)) - 1;
+      m = (mm & low4) | (((mm >> (escape_mask_bits * 4)) & low4)
+                         << (escape_mask_bits * (len - 4)));
+    } else {
+      // Fewer than 4 bytes: at most three table lookups, no need for SIMD.
+      for (size_t k = 0; k < len; k++) {
+        uint8_t c = src[k];
+        if (json_quotable_character[c]) {
+          escape_json_char(char(c), out);
+        } else {
+          *out++ = char(c);
+        }
+      }
+      return size_t(out - initout);
+    }
+    if (m == 0) {
+      copy_lt16(out, src + i, rem);
+      out += rem;
+    } else {
+      out = escape_block(src, out, i, len, m);
+    }
+  }
+  return size_t(out - initout);
+}
+
+#else // SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
 // Writes the escaped version of input to out, returning the number of bytes
 // written. Uses SIMD position finding to locate quotable characters efficiently.
 inline size_t write_string_escaped(const std::string_view input, char *out) {
@@ -42452,9 +51231,13 @@ inline size_t write_string_escaped(const std::string_view input, char *out) {
     escape_json_char(input[location], out);
     location += 1;
   }
-  return out - initout;
+  return size_t(out - initout);
 }

+#endif // SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+#undef SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
+
 simdjson_inline string_builder::string_builder(size_t initial_capacity)
     : buffer(new(std::nothrow) char[initial_capacity]), position(0),
       capacity(buffer.get() != nullptr ? initial_capacity : 0),
@@ -42477,7 +51260,7 @@ simdjson_inline bool string_builder::capacity_check(size_t upcoming_bytes) {
   return is_valid;
 }

-simdjson_inline void string_builder::grow_buffer(size_t desired_capacity) {
+inline void string_builder::grow_buffer(size_t desired_capacity) {
   if (!is_valid) {
     return;
   }
@@ -42533,81 +51316,136 @@ simdjson_inline void string_builder::clear() noexcept {

 namespace internal {

-template <typename number_type, typename = typename std::enable_if<
-                                    std::is_unsigned<number_type>::value>::type>
-simdjson_really_inline int int_log2(number_type x) {
-  return 63 - leading_zeroes(uint64_t(x) | 1);
-}
-
-simdjson_really_inline int fast_digit_count_32(uint32_t x) {
-  static uint64_t table[] = {
-      4294967296,  8589934582,  8589934582,  8589934582,  12884901788,
-      12884901788, 12884901788, 17179868184, 17179868184, 17179868184,
-      21474826480, 21474826480, 21474826480, 21474826480, 25769703776,
-      25769703776, 25769703776, 30063771072, 30063771072, 30063771072,
-      34349738368, 34349738368, 34349738368, 34349738368, 38554705664,
-      38554705664, 38554705664, 41949672960, 41949672960, 41949672960,
-      42949672960, 42949672960};
-  return uint32_t((x + table[int_log2(x)]) >> 32);
-}
-
-simdjson_really_inline int fast_digit_count_64(uint64_t x) {
-  static uint64_t table[] = {9,
-                             99,
-                             999,
-                             9999,
-                             99999,
-                             999999,
-                             9999999,
-                             99999999,
-                             999999999,
-                             9999999999,
-                             99999999999,
-                             999999999999,
-                             9999999999999,
-                             99999999999999,
-                             999999999999999ULL,
-                             9999999999999999ULL,
-                             99999999999999999ULL,
-                             999999999999999999ULL,
-                             9999999999999999999ULL};
-  int y = (19 * int_log2(x) >> 6);
-  y += x > table[y];
-  return y + 1;
-}
-
-template <typename number_type, typename = typename std::enable_if<
-                                    std::is_unsigned<number_type>::value>::type>
-simdjson_really_inline size_t digit_count(number_type v) noexcept {
-  static_assert(sizeof(number_type) == 8 || sizeof(number_type) == 4 ||
-                    sizeof(number_type) == 2 || sizeof(number_type) == 1,
-                "We only support 8-bit, 16-bit, 32-bit and 64-bit numbers");
-  SIMDJSON_IF_CONSTEXPR(sizeof(number_type) <= 4) {
-    return fast_digit_count_32(static_cast<uint32_t>(v));
+// Integer to decimal: James Edward Anhalt III's algorithm
+static const char jeaiii_dd[201] =
+    "00010203040506070809101112131415161718192021222324252627282930313233343536373839"
+    "40414243444546474849505152535455565758596061626364656667686970717273747576777879"
+    "8081828384858687888990919293949596979899";
+static const char jeaiii_fd[201] =
+    "0\0" "1\0" "2\0" "3\0" "4\0" "5\0" "6\0" "7\0" "8\0" "9\0"
+    "10111213141516171819202122232425262728293031323334353637383940414243444546474849"
+    "50515253545556575859606162636465666768697071727374757677787980818283848586878889"
+    "90919293949596979899";
+
+simdjson_really_inline void jeaiii_write_dd(char *p, uint64_t k) noexcept {
+  std::memcpy(p, &jeaiii_dd[2 * k], 2);
+}
+simdjson_really_inline void jeaiii_write_fd(char *p, uint64_t k) noexcept {
+  std::memcpy(p, &jeaiii_fd[2 * k], 2);
+}
+
+// Caller guarantees n < 10^8. Writes 1 to 8 digits.
+simdjson_really_inline char *jeaiii_lt1e8(char *b, uint32_t n) noexcept {
+  constexpr uint64_t mask24 = (uint64_t(1) << 24) - 1;
+  constexpr uint64_t mask32 = (uint64_t(1) << 32) - 1;
+  if (n < 100) {
+    jeaiii_write_fd(b, n);
+    return n < 10 ? b + 1 : b + 2;
+  }
+  if (n < 1000000) {
+    if (n < 10000) {
+      const uint32_t f0 = uint32_t(10 * (1 << 24) / 1e3 + 1) * n;
+      jeaiii_write_fd(b, f0 >> 24);
+      b -= n < 1000;
+      const uint32_t f2 = uint32_t(f0 & mask24) * 100;
+      jeaiii_write_dd(b + 2, f2 >> 24);
+      return b + 4;
+    }
+    const uint64_t f0 = uint64_t(10 * (1ull << 32) / 1e5 + 1) * n;
+    jeaiii_write_fd(b, f0 >> 32);
+    b -= n < 100000;
+    const uint64_t f2 = (f0 & mask32) * 100;
+    jeaiii_write_dd(b + 2, f2 >> 32);
+    const uint64_t f4 = (f2 & mask32) * 100;
+    jeaiii_write_dd(b + 4, f4 >> 32);
+    return b + 6;
+  }
+  const uint64_t f0 = uint64_t(10 * (1ull << 48) / 1e7 + 1) * n >> 16;
+  jeaiii_write_fd(b, f0 >> 32);
+  b -= n < 10000000;
+  const uint64_t f2 = (f0 & mask32) * 100;
+  jeaiii_write_dd(b + 2, f2 >> 32);
+  const uint64_t f4 = (f2 & mask32) * 100;
+  jeaiii_write_dd(b + 4, f4 >> 32);
+  const uint64_t f6 = (f4 & mask32) * 100;
+  jeaiii_write_dd(b + 6, f6 >> 32);
+  return b + 8;
+}
+
+// Caller guarantees z < 10^8. Always writes exactly 8 digits.
+simdjson_really_inline char *jeaiii_8_digits(char *b, uint32_t z) noexcept {
+  constexpr uint64_t mask32 = (uint64_t(1) << 32) - 1;
+  const uint64_t f0 = (uint64_t((1ull << 48) / 1e6 + 1) * z >> 16) + 1;
+  jeaiii_write_dd(b, f0 >> 32);
+  const uint64_t f2 = (f0 & mask32) * 100;
+  jeaiii_write_dd(b + 2, f2 >> 32);
+  const uint64_t f4 = (f2 & mask32) * 100;
+  jeaiii_write_dd(b + 4, f4 >> 32);
+  const uint64_t f6 = (f4 & mask32) * 100;
+  jeaiii_write_dd(b + 6, f6 >> 32);
+  return b + 8;
+}
+
+// Caller guarantees 10^8 <= n < 2^32. Writes 9 or 10 digits.
+simdjson_really_inline char *jeaiii_9_or_10(char *b, uint64_t n) noexcept {
+  constexpr uint64_t mask57 = (uint64_t(1) << 57) - 1;
+  const uint64_t f0 = uint64_t(10 * (1ull << 57) / 1e9 + 1) * n;
+  jeaiii_write_fd(b, f0 >> 57);
+  b -= n < 1000000000;
+  const uint64_t f2 = (f0 & mask57) * 100;
+  jeaiii_write_dd(b + 2, f2 >> 57);
+  const uint64_t f4 = (f2 & mask57) * 100;
+  jeaiii_write_dd(b + 4, f4 >> 57);
+  const uint64_t f6 = (f4 & mask57) * 100;
+  jeaiii_write_dd(b + 6, f6 >> 57);
+  const uint64_t f8 = (f6 & mask57) * 100;
+  jeaiii_write_dd(b + 8, f8 >> 57);
+  return b + 10;
+}
+
+simdjson_really_inline char *write_uint_jeaiii(char *b, uint64_t n) noexcept {
+  if (n < 100000000) {
+    return jeaiii_lt1e8(b, uint32_t(n));
+  }
+  if (n < (uint64_t(1) << 32)) {
+    return jeaiii_9_or_10(b, n);
+  }
+  // At least 10 digits: the low 8 digits, and 2 to 12 digits above them.
+  const uint32_t z = uint32_t(n % 100000000);
+  uint64_t u = n / 100000000;
+  if (u < 100000000) {
+    // u has 2 to 8 digits (if u < 10, n would be below 2^32).
+    b = jeaiii_lt1e8(b, uint32_t(u));
+  } else if (u < (uint64_t(1) << 32)) {
+    b = jeaiii_9_or_10(b, u);
+  } else {
+    // u has 11 or 12 digits: split off 8 more.
+    const uint32_t y = uint32_t(u % 100000000);
+    u /= 100000000;
+    b = jeaiii_lt1e8(b, uint32_t(u)); // 3 or 4 digits
+    b = jeaiii_8_digits(b, y);
   }
-  else {
-    return fast_digit_count_64(static_cast<uint64_t>(v));
-  }
-}
-static const char decimal_table[200] = {
-    0x30, 0x30, 0x30, 0x31, 0x30, 0x32, 0x30, 0x33, 0x30, 0x34, 0x30, 0x35,
-    0x30, 0x36, 0x30, 0x37, 0x30, 0x38, 0x30, 0x39, 0x31, 0x30, 0x31, 0x31,
-    0x31, 0x32, 0x31, 0x33, 0x31, 0x34, 0x31, 0x35, 0x31, 0x36, 0x31, 0x37,
-    0x31, 0x38, 0x31, 0x39, 0x32, 0x30, 0x32, 0x31, 0x32, 0x32, 0x32, 0x33,
-    0x32, 0x34, 0x32, 0x35, 0x32, 0x36, 0x32, 0x37, 0x32, 0x38, 0x32, 0x39,
-    0x33, 0x30, 0x33, 0x31, 0x33, 0x32, 0x33, 0x33, 0x33, 0x34, 0x33, 0x35,
-    0x33, 0x36, 0x33, 0x37, 0x33, 0x38, 0x33, 0x39, 0x34, 0x30, 0x34, 0x31,
-    0x34, 0x32, 0x34, 0x33, 0x34, 0x34, 0x34, 0x35, 0x34, 0x36, 0x34, 0x37,
-    0x34, 0x38, 0x34, 0x39, 0x35, 0x30, 0x35, 0x31, 0x35, 0x32, 0x35, 0x33,
-    0x35, 0x34, 0x35, 0x35, 0x35, 0x36, 0x35, 0x37, 0x35, 0x38, 0x35, 0x39,
-    0x36, 0x30, 0x36, 0x31, 0x36, 0x32, 0x36, 0x33, 0x36, 0x34, 0x36, 0x35,
-    0x36, 0x36, 0x36, 0x37, 0x36, 0x38, 0x36, 0x39, 0x37, 0x30, 0x37, 0x31,
-    0x37, 0x32, 0x37, 0x33, 0x37, 0x34, 0x37, 0x35, 0x37, 0x36, 0x37, 0x37,
-    0x37, 0x38, 0x37, 0x39, 0x38, 0x30, 0x38, 0x31, 0x38, 0x32, 0x38, 0x33,
-    0x38, 0x34, 0x38, 0x35, 0x38, 0x36, 0x38, 0x37, 0x38, 0x38, 0x38, 0x39,
-    0x39, 0x30, 0x39, 0x31, 0x39, 0x32, 0x39, 0x33, 0x39, 0x34, 0x39, 0x35,
-    0x39, 0x36, 0x39, 0x37, 0x39, 0x38, 0x39, 0x39,
-};
+  return jeaiii_8_digits(b, z);
+}
+
+// Writes v at p, which must have to_chars_buffer_size bytes available, and
+// returns the end of what was written.
+simdjson_inline char *write_double(char *p, double v) noexcept {
+#if SIMDJSON_ENABLE_NAN_INF
+  if (simdjson_unlikely(!std::isfinite(v))) {
+    if (std::isnan(v)) {
+      std::memcpy(p, "NaN", 3);
+      return p + 3;
+    }
+    if (v < 0) {
+      *p++ = '-';
+    }
+    std::memcpy(p, "Infinity", 8);
+    return p + 8;
+  }
+#endif
+  return simdjson::internal::to_chars(p, nullptr, v);
+}
 } // namespace internal

 template <typename number_type, typename>
@@ -42635,87 +51473,62 @@ simdjson_inline void string_builder::append(number_type v) noexcept {
     }
   }
   else SIMDJSON_IF_CONSTEXPR(std::is_unsigned<number_type>::value) {
-    // Process 4 digits at a time instead of 2, reducing store operations
-    // and divisions by approximately half for large numbers.
     constexpr size_t max_number_size = 20;
     if (capacity_check(max_number_size)) {
       using unsigned_type = typename std::make_unsigned<number_type>::type;
-      unsigned_type pv = static_cast<unsigned_type>(v);
-      size_t dc = internal::digit_count(pv);
-      char *write_pointer = buffer.get() + position + dc - 1;
-
-      // Process 4 digits per iteration for large numbers
-      while (pv >= 10000) {
-        unsigned_type q = pv / 10000;
-        unsigned_type r = pv % 10000;
-        unsigned_type r_hi = r / 100;  // High 2 digits of remainder
-        unsigned_type r_lo = r % 100;  // Low 2 digits of remainder
-        // Write low 2 digits first (rightmost), then high 2 digits
-        memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
-        memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
-        write_pointer -= 4;
-        pv = q;
-      }
-
-      // Handle remaining 1-4 digits with original 2-digit loop
-      while (pv >= 100) {
-        memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
-        write_pointer -= 2;
-        pv /= 100;
-      }
-      if (pv >= 10) {
-        *write_pointer-- = char('0' + (pv % 10));
-        pv /= 10;
-      }
-      *write_pointer = char('0' + pv);
-      position += dc;
+      char* end = internal::write_uint_jeaiii(
+          buffer.get() + position,
+          static_cast<uint64_t>(static_cast<unsigned_type>(v)));
+      position = end - buffer.get();
     }
   }
   else SIMDJSON_IF_CONSTEXPR(std::is_integral<number_type>::value) {
-    // Same 4-digit batching as unsigned path for signed integers
+    // 19 digits (max abs value of int64_t) + optional minus sign.
     constexpr size_t max_number_size = 20;
     if (capacity_check(max_number_size)) {
       using unsigned_type = typename std::make_unsigned<number_type>::type;
       bool negative = v < 0;
-      unsigned_type pv = static_cast<unsigned_type>(v);
-      if (negative) {
-        pv = 0 - pv; // the 0 is for Microsoft
-      }
-      size_t dc = internal::digit_count(pv);
-      // by always writing the minus sign, we avoid the branch.
+      // 0 - pv (rather than -pv) avoids an MSVC unary-minus warning.
+      unsigned_type pv = negative
+          ? unsigned_type(0) - static_cast<unsigned_type>(v)
+          : static_cast<unsigned_type>(v);
+      // Branchless: always write '-', advance only if negative.
       buffer.get()[position] = '-';
-      position += negative ? 1 : 0;
-      char *write_pointer = buffer.get() + position + dc - 1;
-
-      // Process 4 digits per iteration for large numbers
-      while (pv >= 10000) {
-        unsigned_type q = pv / 10000;
-        unsigned_type r = pv % 10000;
-        unsigned_type r_hi = r / 100;
-        unsigned_type r_lo = r % 100;
-        memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
-        memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
-        write_pointer -= 4;
-        pv = q;
-      }
-
-      // Handle remaining 1-4 digits
-      while (pv >= 100) {
-        memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
-        write_pointer -= 2;
-        pv /= 100;
-      }
-      if (pv >= 10) {
-        *write_pointer-- = char('0' + (pv % 10));
-        pv /= 10;
-      }
-      *write_pointer = char('0' + pv);
-      position += dc;
+      position += negative;
+      char* end = internal::write_uint_jeaiii(
+          buffer.get() + position, static_cast<uint64_t>(pv));
+      position = end - buffer.get();
     }
   }
   else SIMDJSON_IF_CONSTEXPR(std::is_floating_point<number_type>::value) {
-    constexpr size_t max_number_size = 24;
+    // Must reserve to_chars_buffer_size (40): only ~24 chars are emitted,
+    // but to_chars over-writes with fixed-size 16/17-byte copies so the
+    // compiler can inline mem* (see simdjson::internal::to_chars_buffer_size).
+    constexpr size_t max_number_size = simdjson::internal::to_chars_buffer_size;
     if (capacity_check(max_number_size)) {
+#if SIMDJSON_ENABLE_NAN_INF
+      // Check if the input might be NaN or infinity
+      if (simdjson_unlikely(!std::isfinite(v))) {
+        if (std::isnan(v)) {
+          constexpr char nan_literal[] = "NaN";
+          constexpr size_t nan_len = sizeof(nan_literal) - 1;
+
+          std::memcpy(buffer.get() + position, nan_literal, nan_len);
+          position += nan_len;
+        } else {
+          constexpr char inf_literal[] = "Infinity";
+          constexpr size_t inf_len = sizeof(inf_literal) - 1;
+          if (v < 0) {
+            buffer.get()[position] = '-';
+            ++position;
+          }
+          std::memcpy(buffer.get() + position, inf_literal, inf_len);
+          position += inf_len;
+        }
+        return;
+      }
+#endif
+
       // We could specialize for float.
       char *end = simdjson::internal::to_chars(buffer.get() + position, nullptr,
                                                double(v));
@@ -42776,7 +51589,9 @@ simdjson_inline void string_builder::escape_and_append_with_quotes() noexcept {
 #endif

 simdjson_inline void string_builder::append_raw(const char *c) noexcept {
-  size_t len = std::strlen(c);
+  // char_traits::length is constexpr; lets the compiler fold the length
+  // when called with a pointer to a compile-time-constant string.
+  size_t len = std::char_traits<char>::length(c);
   append_raw(c, len);
 }

@@ -42795,6 +51610,14 @@ simdjson_inline void string_builder::append_raw(const char *str,
     position += len;
   }
 }
+
+template <size_t N>
+simdjson_inline void string_builder::append_raw_n(const char *str) noexcept {
+  if (capacity_check(N)) {
+    std::memcpy(buffer.get() + position, str, N);
+    position += N;
+  }
+}
 #if SIMDJSON_SUPPORTS_CONCEPTS
 // Support for optional types (std::optional, etc.)
 template <concepts::optional_type T>
@@ -42824,7 +51647,7 @@ simdjson_inline void string_builder::append(const T &value) {
 #if SIMDJSON_SUPPORTS_RANGES && SIMDJSON_SUPPORTS_CONCEPTS
 // Support for range-based appending (std::ranges::view, etc.)
 template <std::ranges::range R>
-  requires(!std::is_convertible<R, std::string_view>::value && !require_custom_serialization<R>)
+  requires(!std::is_convertible<R, std::string_view>::value && !concepts::optional_type<R> && !require_custom_serialization<R>)
 simdjson_inline void string_builder::append(const R &range) noexcept {
   auto it = std::ranges::begin(range);
   auto end = std::ranges::end(range);
@@ -43169,16 +51992,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
 }
 #endif

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
-                                uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
-  return _addcarry_u64(0, value1, value2,
-                       reinterpret_cast<unsigned __int64 *>(result));
-#else
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-#endif
-}

 } // unnamed namespace
 } // namespace haswell
@@ -43917,7 +52730,7 @@ public:
 #if SIMDJSON_SUPPORTS_RANGES && SIMDJSON_SUPPORTS_CONCEPTS
   // Support for range-based appending (std::ranges::view, etc.)
   template <std::ranges::range R>
-requires (!std::is_convertible<R, std::string_view>::value && !require_custom_serialization<R>)
+requires (!std::is_convertible<R, std::string_view>::value && !concepts::optional_type<R> && !require_custom_serialization<R>)
   simdjson_inline void append(const R &range) noexcept;
 #endif
   /**
@@ -43931,6 +52744,15 @@ requires (!std::is_convertible<R, std::string_view>::value && !require_custom_se
    * There is no UTF-8 validation.
    */
   simdjson_inline void append_raw(const char *str, size_t len) noexcept;
+
+  /**
+   * Append exactly N characters from str. The length is a template parameter
+   * so the compiler can fully inline the memcpy with a compile-time-constant
+   * size, avoiding the libc call. Used for compile-time-constant keys in the
+   * reflection struct atom.
+   */
+  template <size_t N>
+  simdjson_inline void append_raw_n(const char *str) noexcept;
 #if SIMDJSON_EXCEPTIONS
   /**
    * Creates an std::string from the written JSON buffer.
@@ -43982,6 +52804,26 @@ requires (!std::is_convertible<R, std::string_view>::value && !require_custom_se
    */
   simdjson_inline size_t size() const noexcept;

+  // ============================================================
+  // Internal hooks for the position-as-local writer in json_builder.h.
+  // These exist so the reflection atom code can hold buffer pointer,
+  // position and capacity in registers across long write chains rather
+  // than reloading them after every char* write (strict aliasing
+  // forces those reloads when accessed via members of *this). User
+  // code should NOT call these directly.
+  // ============================================================
+  simdjson_inline char *unsafe_data() noexcept { return buffer.get(); }
+  simdjson_inline size_t unsafe_position() const noexcept { return position; }
+  simdjson_inline size_t unsafe_capacity() const noexcept { return capacity; }
+  simdjson_inline void unsafe_set_position(size_t p) noexcept { position = p; }
+  /// Make capacity available for at least `n` more bytes after the current
+  /// position. Returns false if the allocation failed.
+  simdjson_inline bool unsafe_grow(size_t needed_total_capacity) noexcept {
+    grow_buffer(needed_total_capacity);
+    return is_valid;
+  }
+  simdjson_inline bool unsafe_is_valid() const noexcept { return is_valid; }
+
 private:
   /**
    * Returns true if we can write at least upcoming_bytes bytes.
@@ -43995,7 +52837,7 @@ private:
    * If the allocation fails, is_valid is set to false. We expect
    * that this function would not be repeatedly called.
    */
-  simdjson_inline void grow_buffer(size_t desired_capacity);
+  inline void grow_buffer(size_t desired_capacity);

   /**
    * We use this helper function to make sure that is_valid is kept consistent.
@@ -44052,6 +52894,7 @@ simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initi
 /* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_STRING_BUILDER_H */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/builder/json_string_builder.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/concepts.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/annotations.h" */
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
 #if SIMDJSON_STATIC_REFLECTION

@@ -44069,64 +52912,370 @@ namespace simdjson {
 namespace haswell {
 namespace builder {

-template <class T>
+// Forward-declare helpers defined in json_string_builder-inl.h so the
+// writer-based atom code below can call them (the -inl.h is not yet
+// included at the point this header is parsed; without these forwards,
+// name lookup falls back to the wrong outer namespace).
+namespace internal {
+simdjson_really_inline char *write_uint_jeaiii(char *p, uint64_t v) noexcept;
+simdjson_inline char *write_double(char *p, double v) noexcept;
+} // namespace internal
+simdjson_really_inline size_t write_string_escaped(const std::string_view input, char *out);
+
+SIMDJSON_PUSH_DISABLE_WARNINGS
+SIMDJSON_DISABLE_GCC_WARNING(-Warray-bounds)
+#if !defined(__clang__)
+SIMDJSON_DISABLE_GCC_WARNING(-Wstringop-overflow)
+#endif
+
+// =============================================================
+// `writer`: position-as-local hot-path writer used by the reflection
+// atom code below. Holds the buffer pointer, write position and
+// capacity in three fields that, once `writer` itself is a stack-local
+// in the caller and all atom() functions are inlined, become true
+// register-resident locals after SROA. Glaze achieves the same effect
+// by passing `B&& b, auto&& ix` through every helper. Holding `pos`
+// in a register (rather than as a member of string_builder) is what
+// breaks the strict-aliasing penalty on every char* write through the
+// buffer, which forces a reload of `b.position` and `b.capacity`
+// after every byte.
+//
+// basic_writer<false> (below) is the unchecked variant: the caller has
+// already reserved enough capacity for everything the write chain can
+// produce (see bound_detail::size_bound), so ensure() compiles away.
+// =============================================================
+template <bool Checked>
+struct basic_writer {
+  static constexpr bool checked = Checked;
+  char *ptr;        // buffer pointer (refreshed after a grow)
+  size_t pos;       // write position (local)
+  size_t cap;       // capacity (refreshed after a grow)
+  string_builder &sb;  // back-ref for grow / sync
+
+  // Snapshot string_builder state into a writer for the duration of
+  // a write chain.
+  simdjson_really_inline basic_writer(string_builder &builder) noexcept
+      : ptr(builder.unsafe_data())
+      , pos(builder.unsafe_position())
+      , cap(builder.unsafe_capacity())
+      , sb(builder) {}
+
+  // Write the local position back to the underlying string_builder.
+  // Caller is responsible for invoking before the writer is dropped
+  // (otherwise data is lost). Idempotent.
+  simdjson_really_inline void sync() noexcept {
+    sb.unsafe_set_position(pos);
+  }
+
+  // Ensure at least `n` more bytes of free capacity. Grows the
+  // underlying buffer if needed (rare path). Returns false on
+  // allocation failure.
+  simdjson_really_inline bool ensure(size_t n) noexcept {
+    // pos <= cap, and cap is the size of a live allocation, so pos + n
+    // cannot wrap when n is a small constant or a compile-time length.
+    // Callers passing a size derived from input (the string atoms) must
+    // bound it against pos themselves. Keep the `pos + n <= cap` form:
+    // `n <= cap - pos` is measurably slower once the serializer is inlined.
+    if (simdjson_likely(pos + n <= cap)) { return true; }
+    return grow_slow(n);
+  }
+
+  simdjson_never_inline bool grow_slow(size_t n) noexcept {
+    // Detect overflow.
+    // This is pedantic except maybe on 32-bit targets.
+    if (simdjson_unlikely(pos + n < pos)) return false;
+    sb.unsafe_set_position(pos);
+    // even if 2*capacity overflows, the (std::max) below will pick the needed value,
+    // so we do not need a separate overflow check here.
+    if (!sb.unsafe_grow((std::max)(cap * 2, pos + n))) {
+      // The string_builder freed its buffer and is now invalid (null buffer,
+      // zero capacity and position). Mirror that state so that every later
+      // ensure() fails too: callers only return from the current atom, and
+      // their callers keep writing.
+      ptr = nullptr;
+      pos = 0;
+      cap = 0;
+      return false;
+    }
+    ptr = sb.unsafe_data();
+    cap = sb.unsafe_capacity();
+    return true;
+  }
+};
+
+// The unchecked writer writes into a raw buffer that the caller sized with
+// serialized_size_bound: it never grows and needs no string_builder.
+template <>
+struct basic_writer<false> {
+  static constexpr bool checked = false;
+  char *ptr;
+  size_t pos;
+
+  simdjson_really_inline basic_writer(char *buffer, size_t position) noexcept
+      : ptr(buffer), pos(position) {}
+
+  simdjson_really_inline bool ensure(size_t) const noexcept { return true; }
+};
+
+using writer = basic_writer<true>;
+using unchecked_writer = basic_writer<false>;
+
+// Bytes reserved past the size bound for an unchecked writer: it may then
+// write a little past the end of what it produces (e.g., copy keys as whole
+// 16-byte blocks).
+inline constexpr size_t unchecked_slack = 64;
+
+consteval size_t padded_key_length(size_t length) {
+  return (length + 15) / 16 * 16;
+}
+
+// === Helper: invoke a string_builder member that writes variable-length
+// content (escape_and_append_with_quotes etc), syncing the writer's local
+// state before the call and reloading after. Used for string fields where
+// rewriting the entire SIMD escape path through the writer would be a much
+// bigger refactor. f may be user code (a with<Adapter> serializer) that
+// throws: the exception then propagates to the caller.
+template <class W, class F>
+simdjson_really_inline void call_through_string_builder(W &w, F &&f) noexcept(noexcept(f(w.sb))) {
+  w.sync();
+  f(w.sb);
+  w.ptr = w.sb.unsafe_data();
+  w.pos = w.sb.unsafe_position();
+  w.cap = w.sb.unsafe_capacity();
+}
+
+// Helpers implementing the serialization side of the annotations (see
+// simdjson/annotations.h) for reflected structures.
+namespace annotation_detail {
+
+// A member is serialized unless it is annotated with skip or skip_serializing.
+consteval bool is_serialized_member(std::meta::info dm) {
+  return !simdjson::detail::has_annotation(dm, ^^simdjson::detail::skip_tag)
+      && !simdjson::detail::has_annotation(dm, ^^simdjson::detail::skip_serializing_tag);
+}
+
+// False when the member has a skip_serializing_if<pred> annotation and
+// pred(value) is true.
+template <auto dm, typename V>
+simdjson_really_inline bool should_serialize(const V &value) {
+  constexpr std::meta::info skip_if_type = simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::skip_serializing_if_t);
+  if constexpr (skip_if_type != std::meta::info{}) {
+    using skip_if = typename [: skip_if_type :];
+    return !skip_if::predicate(value);
+  } else {
+    (void)value;
+    return true;
+  }
+}
+
+// Serialize a member value, through its with<Adapter> annotation when the
+// adapter provides a serialize function.
+template <auto dm, class W, typename V>
+simdjson_really_inline void atom_member(W &w, const V &value) {
+  constexpr std::meta::info with_type = simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::with_t);
+  if constexpr (with_type != std::meta::info{}) {
+    using adapter = typename [: with_type :]::adapter;
+    if constexpr (requires(string_builder &b) { adapter::serialize(b, value); }) {
+      call_through_string_builder(w, [&](string_builder &b) { adapter::serialize(b, value); });
+    } else {
+      atom(w, value);
+    }
+  } else {
+    atom(w, value);
+  }
+}
+
+// Write the "key":value pairs of the members of t (without the braces), each
+// preceded by a comma unless it is the first one. The members of a member
+// annotated with flatten are written in its place.
+template <class W, class T>
+simdjson_really_inline void atom_fields(W &w, const T &t, bool &first) {
+  // Per-field block: ensure key+value worst case, then write key + value
+  // through the writer's local pos. For arithmetic fields, the integer
+  // write happens directly via write_uint_jeaiii on w.ptr+w.pos, so pos
+  // never round-trips through memory.
+  template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+    if constexpr (is_serialized_member(dm)) {
+      if (should_serialize<dm>(t.[:dm:])) {
+        if constexpr (simdjson::detail::has_annotation(dm, ^^simdjson::detail::flatten_tag)) {
+          static_assert(std::meta::is_class_type(simdjson::detail::flattened_type(dm)));
+          using flattened = std::remove_cvref_t<decltype(t.[:dm:])>;
+          static_assert(!concepts::container_but_not_string<flattened> && !concepts::string_view_keyed_map<flattened> &&
+                        !concepts::appendable_containers<flattened> && !concepts::optional_type<flattened> &&
+                        !concepts::smart_pointer<flattened> && !std::is_same_v<flattened, std::string> &&
+                        !std::is_same_v<flattened, std::string_view> && !require_custom_serialization<flattened>,
+                        "simdjson::flatten requires a member whose type is a structure serialized member by member");
+          atom_fields(w, t.[:dm:], first);
+        } else {
+          // Copy the key as whole 16-byte blocks from a zero-padded copy (one
+          // load and one store); ensure() reserves the padded length, and the
+          // unchecked writer has slack past its bound. Prior related work:
+          // jsonifier copies a power-of-two padded key and advances the cursor
+          // by the real length (serialize_impl.hpp, packed_blitter,
+          // https://github.com/nihilai-collective/Jsonifier).
+          constexpr const char* key_name = simdjson::get_json_key_name<dm>();
+          constexpr size_t first_key_len = constevalutil::consteval_to_quoted_escaped(key_name).size() + 1;
+          constexpr size_t rest_key_len = first_key_len + 1;
+          constexpr auto first_key = std::define_static_string(
+              constevalutil::consteval_to_quoted_escaped(key_name) + ":" +
+              std::string(padded_key_length(first_key_len) - first_key_len, '\0'));
+          constexpr auto rest_key = std::define_static_string(
+              std::string(",") + constevalutil::consteval_to_quoted_escaped(key_name) + ":" +
+              std::string(padded_key_length(rest_key_len) - rest_key_len, '\0'));
+          if (!w.ensure(padded_key_length(rest_key_len))) { return; }
+          if (first) {
+            std::memcpy(w.ptr + w.pos, first_key, padded_key_length(first_key_len));
+            w.pos += first_key_len;
+          } else {
+            std::memcpy(w.ptr + w.pos, rest_key, padded_key_length(rest_key_len));
+            w.pos += rest_key_len;
+          }
+          first = false;
+          atom_member<dm>(w, t.[:dm:]);
+        }
+      }
+    }
+  };
+}
+
+} // namespace annotation_detail
+
+template <class W, class T>
   requires(concepts::container_but_not_string<T> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &t) {
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
   auto it = t.begin();
   auto end = t.end();
   if (it == end) {
-    b.append_raw("[]");
+    if (!w.ensure(2)) return;
+    std::memcpy(w.ptr + w.pos, "[]", 2);
+    w.pos += 2;
     return;
   }
-  b.append('[');
-  atom(b, *it);
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = '[';
+  atom(w, *it);
   ++it;
   for (; it != end; ++it) {
-    b.append(',');
-    atom(b, *it);
+    if (!w.ensure(1)) return;
+    w.ptr[w.pos++] = ',';
+    atom(w, *it);
   }
-  b.append(']');
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = ']';
 }

-template <class T>
+template <class W, class T>
   requires(std::is_same_v<T, std::string> ||
            std::is_same_v<T, std::string_view> ||
            std::is_same_v<T, const char *> ||
            std::is_same_v<T, char>)
-constexpr void atom(string_builder &b, const T &t) {
-  b.escape_and_append_with_quotes(t);
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
+  // Inline the escape path through the writer so we never round-trip
+  // pos through memory for string fields (Twitter is dominated by
+  // these -- sync/reload around each string was a real cost).
+  std::string_view input;
+  if constexpr (std::is_same_v<T, char>) {
+    input = std::string_view(&t, 1);
+  } else {
+    input = std::string_view(t);
+  }
+  // Worst-case escape: every byte expands to \uXXXX (6 chars), plus 2 quotes.
+  // Guard against w.pos + 2 + 6 * input.size() wrapping for huge inputs -- if
+  // it wrapped to a small value, ensure() would spuriously succeed and the
+  // subsequent escape would overflow the buffer. max - w.pos cannot wrap, and
+  // size < (max - pos) / 6 implies pos + 6 * size + 6 <= max.
+  // Note that this is pedantic except maybe on 32-bit targets.
+  if constexpr (W::checked) {
+    if (simdjson_unlikely(input.size() >= ((std::numeric_limits<size_t>::max)() - w.pos) / 6)) { return; }
+    if (!w.ensure(2 + 6 * input.size())) { return; }
+  }
+  w.ptr[w.pos++] = '"';
+  w.pos += write_string_escaped(input, w.ptr + w.pos);
+  w.ptr[w.pos++] = '"';
 }

-template <concepts::string_view_keyed_map T>
+template <class W, concepts::string_view_keyed_map T>
   requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &m) {
+simdjson_really_inline constexpr void atom(W &w, const T &m) {
   if (m.empty()) {
-    b.append_raw("{}");
+    if (!w.ensure(2)) return;
+    std::memcpy(w.ptr + w.pos, "{}", 2);
+    w.pos += 2;
     return;
   }
-  b.append('{');
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = '{';
   bool first = true;
   for (const auto& [key, value] : m) {
     if (!first) {
-      b.append(',');
+      if (!w.ensure(1)) return;
+      w.ptr[w.pos++] = ',';
     }
     first = false;
-    // Keys must be convertible to string_view per the concept
-    b.escape_and_append_with_quotes(key);
-    b.append(':');
-    atom(b, value);
+    // Keys must be convertible to string_view per the concept.
+    std::string_view key_sv(key);
+    // Guard against w.pos + 3 + 6 * key_sv.size() wrapping for huge keys, if
+    // it wrapped to a small value, ensure() would spuriously succeed and the
+    // subsequent escape would overflow the buffer. max - w.pos cannot wrap, and
+    // size < (max - pos) / 6 implies pos + 6 * size + 6 <= max.
+    // Note that this is pedantic except maybe on 32-bit targets.
+    if constexpr (W::checked) {
+      if (simdjson_unlikely(key_sv.size() >= ((std::numeric_limits<size_t>::max)() - w.pos) / 6)) { return; }
+      if (!w.ensure(2 + 6 * key_sv.size() + 1)) { return; }
+    }
+    w.ptr[w.pos++] = '"';
+    w.pos += write_string_escaped(key_sv, w.ptr + w.pos);
+    w.ptr[w.pos++] = '"';
+    w.ptr[w.pos++] = ':';
+    atom(w, value);
   }
-  b.append('}');
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = '}';
 }


-template<typename number_type,
+template<class W, typename number_type,
          typename = typename std::enable_if<std::is_arithmetic<number_type>::value && !std::is_same_v<number_type, char>>::type>
-constexpr void atom(string_builder &b, const number_type t) {
-  b.append(t);
+simdjson_really_inline constexpr void atom(W &w, const number_type t) {
+  // Booleans / floats: defer to string_builder (rare path; keeps writer hot
+  // path free of float-formatter machinery). For integers, write directly
+  // via jeaiii using local pos.
+  if constexpr (std::is_same_v<number_type, bool>) {
+    if (t) {
+      if (!w.ensure(4)) return;
+      std::memcpy(w.ptr + w.pos, "true", 4);
+      w.pos += 4;
+    } else {
+      if (!w.ensure(5)) return;
+      std::memcpy(w.ptr + w.pos, "false", 5);
+      w.pos += 5;
+    }
+  } else if constexpr (std::is_floating_point_v<number_type>) {
+    if constexpr (W::checked) {
+      call_through_string_builder(w, [&](string_builder &b) { b.append(t); });
+    } else {
+      w.pos = size_t(internal::write_double(w.ptr + w.pos, double(t)) - w.ptr);
+    }
+  } else if constexpr (std::is_unsigned_v<number_type>) {
+    if (!w.ensure(20)) return;
+    char *end = internal::write_uint_jeaiii(
+        w.ptr + w.pos, static_cast<uint64_t>(t));
+    w.pos = static_cast<size_t>(end - w.ptr);
+  } else {
+    // signed integral
+    if (!w.ensure(20)) return;
+    using U = typename std::make_unsigned<number_type>::type;
+    bool negative = t < 0;
+    U pv = negative ? U(0) - static_cast<U>(t) : static_cast<U>(t);
+    w.ptr[w.pos] = '-';
+    w.pos += negative;
+    char *end = internal::write_uint_jeaiii(
+        w.ptr + w.pos, static_cast<uint64_t>(pv));
+    w.pos = static_cast<size_t>(end - w.ptr);
+  }
 }

-template <class T>
+template <class W, class T>
   requires(std::is_class_v<T> && !concepts::container_but_not_string<T> &&
            !concepts::string_view_keyed_map<T> &&
            !concepts::optional_type<T> &&
@@ -44136,92 +53285,259 @@ template <class T>
            !std::is_same_v<T, std::string_view> &&
            !std::is_same_v<T, const char*> &&
            !std::is_same_v<T, char> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &t) {
-  int i = 0;
-  b.append('{');
-  template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
-    if (i != 0)
-      b.append(',');
-    constexpr auto key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(dm)));
-    b.append_raw(key);
-    b.append(':');
-    atom(b, t.[:dm:]);
-    i++;
-  };
-  b.append('}');
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
+  if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag)) {
+    // A transparent structure is serialized as its single member.
+    constexpr auto dm = simdjson::detail::transparent_member(^^T);
+    annotation_detail::atom_member<dm>(w, t.[:dm:]);
+  } else {
+    bool first = true;
+    if (!w.ensure(1)) { return; }
+    w.ptr[w.pos++] = '{';
+    annotation_detail::atom_fields(w, t, first);
+    if (!w.ensure(1)) { return; }
+    w.ptr[w.pos++] = '}';
+  }
 }

 // Support for optional types (std::optional, etc.)
-template <concepts::optional_type T>
+template <class W, concepts::optional_type T>
   requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &opt) {
+simdjson_really_inline constexpr void atom(W &w, const T &opt) {
   if (opt) {
-    atom(b, opt.value());
+    atom(w, opt.value());
   } else {
-    b.append_raw("null");
+    if (!w.ensure(4)) return;
+    std::memcpy(w.ptr + w.pos, "null", 4);
+    w.pos += 4;
   }
 }

 // Support for smart pointers (std::unique_ptr, std::shared_ptr, etc.)
-template <concepts::smart_pointer T>
+template <class W, concepts::smart_pointer T>
   requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &ptr) {
+simdjson_really_inline constexpr void atom(W &w, const T &ptr) {
   if (ptr) {
-    atom(b, *ptr);
+    atom(w, *ptr);
   } else {
-    b.append_raw("null");
+    if (!w.ensure(4)) return;
+    std::memcpy(w.ptr + w.pos, "null", 4);
+    w.pos += 4;
   }
 }

 // Support for enums - serialize as string representation using expand approach from P2996R12
-template <typename T>
+template <class W, typename T>
   requires(std::is_enum_v<T> && !require_custom_serialization<T>)
-void atom(string_builder &b, const T &e) {
+simdjson_really_inline void atom(W &w, const T &e) {
 #if SIMDJSON_STATIC_REFLECTION
-  constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
+  static constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
   template for (constexpr auto enum_val : enumerators) {
-    constexpr auto enum_str = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(enum_val)));
+    constexpr auto enum_str = std::define_static_string(constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<enum_val>()));
+    constexpr size_t enum_str_len = std::char_traits<char>::length(enum_str);
     if (e == [:enum_val:]) {
-      b.append_raw(enum_str);
+      if (!w.ensure(enum_str_len)) return;
+      std::memcpy(w.ptr + w.pos, enum_str, enum_str_len);
+      w.pos += enum_str_len;
       return;
     }
   };
   // Fallback to integer if enum value not found
-  atom(b, static_cast<std::underlying_type_t<T>>(e));
+  atom(w, static_cast<std::underlying_type_t<T>>(e));
 #else
   // Fallback: serialize as integer if reflection not available
-  atom(b, static_cast<std::underlying_type_t<T>>(e));
+  atom(w, static_cast<std::underlying_type_t<T>>(e));
 #endif
 }

 // Support for appendable containers that don't have operator[] (sets, etc.)
-template <concepts::appendable_containers T>
+template <class W, concepts::appendable_containers T>
   requires(!concepts::container_but_not_string<T> && !concepts::string_view_keyed_map<T> &&
            !concepts::optional_type<T> && !concepts::smart_pointer<T> &&
            !std::is_same_v<T, std::string> &&
            !std::is_same_v<T, std::string_view> && !std::is_same_v<T, const char*> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &container) {
+simdjson_really_inline constexpr void atom(W &w, const T &container) {
   if (container.empty()) {
-    b.append_raw("[]");
+    if (!w.ensure(2)) return;
+    std::memcpy(w.ptr + w.pos, "[]", 2);
+    w.pos += 2;
     return;
   }
-  b.append('[');
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = '[';
   bool first = true;
   for (const auto& item : container) {
     if (!first) {
-      b.append(',');
+      if (!w.ensure(1)) return;
+      w.ptr[w.pos++] = ',';
     }
     first = false;
-    atom(b, item);
+    atom(w, item);
+  }
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = ']';
+}
+
+// =============================================================
+// Size bound: an upper bound on the number of bytes that atom(w, t) writes.
+// Computing it first lets append() reserve the capacity once and then run
+// the whole write chain through an unchecked_writer, without a capacity
+// check before every write. It mirrors the atom() overloads above.
+//
+// Prior related work: jsonifier sizes the document first, counting 6 bytes
+// per string byte, resizes once to that bound plus slack, and writes with
+// no capacity check on each store. to_json() does that resize through
+// std::string::resize_and_overwrite. See serializer.hpp and
+// serialize_impl.hpp in https://github.com/nihilai-collective/Jsonifier
+// and https://nihilai-collective.net/serialization.
+// =============================================================
+namespace bound_detail {
+
+// Whether size_bound covers everything that atom() writes for T: not when a
+// member is serialized by a with<Adapter> serializer, which writes an unknown
+// amount through the string_builder.
+template <class T>
+consteval bool is_bounded() {
+  if constexpr (require_custom_serialization<T>) {
+    return false;
+  } else if constexpr (std::is_same_v<T, std::string> || std::is_same_v<T, std::string_view> ||
+                       std::is_same_v<T, const char *> || std::is_arithmetic_v<T> || std::is_enum_v<T>) {
+    return true;
+  } else if constexpr (concepts::optional_type<T> || concepts::smart_pointer<T>) {
+    return is_bounded<std::remove_cvref_t<decltype(*std::declval<const T &>())>>();
+  } else if constexpr (concepts::string_view_keyed_map<T>) {
+    return is_bounded<std::remove_cvref_t<typename T::mapped_type>>();
+  } else if constexpr (concepts::container_but_not_string<T> || concepts::appendable_containers<T>) {
+    return is_bounded<std::remove_cvref_t<std::ranges::range_value_t<T>>>();
+  } else {
+    bool bounded = true;
+    template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+      if constexpr (annotation_detail::is_serialized_member(dm)) {
+        bounded = bounded && simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::with_t) == std::meta::info{} &&
+                  is_bounded<std::remove_cvref_t<decltype(std::declval<const T &>().[:dm:])>>();
+      }
+    };
+    return bounded;
   }
-  b.append(']');
 }

-// append functions that delegate to atom functions for primitive types
+template <class T>
+consteval size_t enum_bound() {
+  size_t bound = 20; // the integer fallback
+  template for (constexpr auto enum_val : std::define_static_array(std::meta::enumerators_of(^^T))) {
+    constexpr size_t len = std::char_traits<char>::length(std::define_static_string(
+        constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<enum_val>())));
+    bound = (std::max)(bound, len);
+  };
+  return bound;
+}
+
+template <class T>
+simdjson_really_inline size_t size_bound(const T &t) noexcept;
+
+// Bound for the "key":value pairs of a structure, commas included.
+template <class T>
+simdjson_really_inline size_t fields_bound(const T &t) noexcept {
+  size_t bound = 0;
+  template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+    if constexpr (annotation_detail::is_serialized_member(dm)) {
+      if constexpr (simdjson::detail::has_annotation(dm, ^^simdjson::detail::flatten_tag)) {
+        bound += fields_bound(t.[:dm:]);
+      } else {
+        constexpr size_t rest_key_len = constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<dm>()).size() + 2;
+        bound += rest_key_len + size_bound(t.[:dm:]);
+      }
+    }
+  };
+  return bound;
+}
+
+template <class T>
+simdjson_really_inline size_t size_bound([[maybe_unused]] const T &t) noexcept {
+  if constexpr (std::is_same_v<T, char>) {
+    return 2 + 6;
+  } else if constexpr (std::is_same_v<T, std::string> || std::is_same_v<T, std::string_view> ||
+                       std::is_same_v<T, const char *>) {
+    // Every byte may become \uXXXX, plus the quotes.
+    return 2 + 6 * std::string_view(t).size();
+  } else if constexpr (std::is_same_v<T, bool>) {
+    return 5;
+  } else if constexpr (std::is_floating_point_v<T>) {
+    return simdjson::internal::to_chars_buffer_size;
+  } else if constexpr (std::is_arithmetic_v<T>) {
+    return 20;
+  } else if constexpr (std::is_enum_v<T>) {
+    return enum_bound<T>();
+  } else if constexpr (concepts::optional_type<T> || concepts::smart_pointer<T>) {
+    return t ? size_bound(*t) : 4;
+  } else if constexpr (concepts::string_view_keyed_map<T>) {
+    size_t bound = 2;
+    for (const auto &[key, value] : t) {
+      // comma, quotes, colon
+      bound += 4 + 6 * std::string_view(key).size() + size_bound(value);
+    }
+    return bound;
+  } else if constexpr (concepts::container_but_not_string<T> || concepts::appendable_containers<T>) {
+    using value_type = std::remove_cvref_t<std::ranges::range_value_t<T>>;
+    if constexpr (std::is_arithmetic_v<value_type> && !std::is_same_v<value_type, char>) {
+      // A fixed bound per element: no need to visit them.
+      return 2 + size_t(std::ranges::distance(t)) * (1 + size_bound(value_type{}));
+    } else {
+      size_t bound = 2;
+#if defined(__GNUC__) && !defined(__clang__)
+#pragma GCC novector // a vector loop is slower on short containers
+#endif
+#pragma GCC unroll 4
+      for (const auto &item : t) {
+        bound += 1 + size_bound(item);
+      }
+      return bound;
+    }
+  } else if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag)) {
+    constexpr auto dm = simdjson::detail::transparent_member(^^T);
+    return size_bound(t.[:dm:]);
+  } else {
+    return 2 + fields_bound(t);
+  }
+}
+
+} // namespace bound_detail
+
+// Write t through an unchecked writer when its size bound is available,
+// reserving that many bytes first, and through the checked writer otherwise.
+template <class T>
+simdjson_really_inline void append_bounded(string_builder &b, const T &t) {
+  // On 32-bit systems, the bound could overflow: keep the checked writer.
+  if constexpr (sizeof(size_t) >= 8 && bound_detail::is_bounded<T>()) {
+    const size_t bound = bound_detail::size_bound(t) + unchecked_slack;
+    const size_t pos = b.unsafe_position();
+    // The bound is a sum of in-memory sizes times a small constant: it cannot
+    // overflow on a 64-bit system. Be pedantic elsewhere.
+    if (sizeof(size_t) >= 8 || bound <= (std::numeric_limits<size_t>::max)() - pos) {
+      const size_t cap = b.unsafe_capacity();
+      // Grow geometrically so that many small appends stay amortized.
+      if (pos + bound <= cap || b.unsafe_grow((std::max)(cap * 2, pos + bound))) {
+        unchecked_writer w(b.unsafe_data(), pos);
+        atom(w, t);
+        b.unsafe_set_position(w.pos);
+      }
+      return;
+    }
+  }
+  writer w(b);
+  atom(w, t);
+  w.sync();
+}
+
+// append() -- top-level entry. Each overload constructs a stack-local
+// writer, runs atom(w, t) through the inlined call chain, then syncs
+// the local position back into the string_builder.
 template <class T>
   requires(std::is_arithmetic_v<T> && !std::is_same_v<T, char>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  writer w(b);
+  atom(w, t);
+  w.sync();
 }

 template <class T>
@@ -44229,20 +53545,22 @@ template <class T>
            std::is_same_v<T, std::string_view> ||
            std::is_same_v<T, const char *> ||
            std::is_same_v<T, char>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  writer w(b);
+  atom(w, t);
+  w.sync();
 }

 template <concepts::optional_type T>
   requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  append_bounded(b, t);
 }

 template <concepts::smart_pointer T>
   requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  append_bounded(b, t);
 }

 template <concepts::appendable_containers T>
@@ -44250,14 +53568,14 @@ template <concepts::appendable_containers T>
            !concepts::optional_type<T> && !concepts::smart_pointer<T> &&
            !std::is_same_v<T, std::string> &&
            !std::is_same_v<T, std::string_view> && !std::is_same_v<T, const char*> && !require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  append_bounded(b, t);
 }

 template <concepts::string_view_keyed_map T>
   requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  append_bounded(b, t);
 }

 // works for struct
@@ -44271,39 +53589,15 @@ template <class Z>
            !std::is_same_v<Z, std::string_view> &&
            !std::is_same_v<Z, const char*> &&
            !std::is_same_v<Z, char> && !require_custom_serialization<Z>)
-void append(string_builder &b, const Z &z) {
-  int i = 0;
-  b.append('{');
-  template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^Z, std::meta::access_context::unchecked()))) {
-    if (i != 0)
-      b.append(',');
-    constexpr auto key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(dm)));
-    b.append_raw(key);
-    b.append(':');
-    atom(b, z.[:dm:]);
-    i++;
-  };
-  b.append('}');
+simdjson_inline void append(string_builder &b, const Z &z) {
+  append_bounded(b, z);
 }

 // works for container that have begin() and end() iterators
 template <class Z>
   requires(concepts::container_but_not_string<Z> && !require_custom_serialization<Z>)
-void append(string_builder &b, const Z &z) {
-  auto it = z.begin();
-  auto end = z.end();
-  if (it == end) {
-    b.append_raw("[]");
-    return;
-  }
-  b.append('[');
-  atom(b, *it);
-  ++it;
-  for (; it != end; ++it) {
-    b.append(',');
-    atom(b, *it);
-  }
-  b.append(']');
+simdjson_inline void append(string_builder &b, const Z &z) {
+  append_bounded(b, z);
 }

 template <class Z>
@@ -44314,22 +53608,40 @@ void append(string_builder &b, const Z &z) {


 template <class Z>
-simdjson_warn_unused simdjson_result<std::string> to_json_string(const Z &z, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
-  string_builder b(initial_capacity);
-  append(b, z);
-  std::string_view s;
-  if(auto e = b.view().get(s); e) { return e; }
-  return std::string(s);
+simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
+  if constexpr (sizeof(size_t) >= 8 && bound_detail::is_bounded<Z>()) {
+    // Write straight into s, sized by the bound: no intermediate buffer, no copy.
+    // Prior related work: jsonifier's serializeJson resizes once through
+    // resize_and_overwrite (serializer.hpp).
+    (void)initial_capacity;
+    const size_t bound = bound_detail::size_bound(z) + unchecked_slack;
+    auto write = [&z](char *p) noexcept {
+      unchecked_writer w(p, 0);
+      atom(w, z);
+      return w.pos;
+    };
+#if defined(__cpp_lib_string_resize_and_overwrite) && __cpp_lib_string_resize_and_overwrite >= 202110L
+    s.resize_and_overwrite(bound, [&write](char *p, size_t) noexcept { return write(p); });
+#else
+    s.resize(bound);
+    s.resize(write(s.data()));
+#endif
+    return SUCCESS;
+  } else {
+    string_builder b(initial_capacity);
+    append(b, z);
+    std::string_view view;
+    if(auto e = b.view().get(view); e) { return e; }
+    s.assign(view);
+    return SUCCESS;
+  }
 }

 template <class Z>
-simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
-  string_builder b(initial_capacity);
-  append(b, z);
-  std::string_view view;
-  if(auto e = b.view().get(view); e) { return e; }
-  s.assign(view);
-  return SUCCESS;
+simdjson_warn_unused simdjson_result<std::string> to_json_string(const Z &z, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
+  std::string s;
+  if(auto e = to_json(z, s, initial_capacity); e) { return e; }
+  return s;
 }

 template <class Z>
@@ -44342,40 +53654,41 @@ string_builder& operator<<(string_builder& b, const Z& z) {
 template<constevalutil::fixed_string... FieldNames, typename T>
   requires(std::is_class_v<T> && (sizeof...(FieldNames) > 0))
 void extract_from(string_builder &b, const T &obj) {
-  // Helper to check if a field name matches any of the requested fields
-  auto should_extract = [](std::string_view field_name) constexpr -> bool {
-    return ((FieldNames.view() == field_name) || ...);
-  };
-
-  b.append('{');
+  writer w(b);
+  if (!w.ensure(1)) { w.sync(); return; }
+  w.ptr[w.pos++] = '{';
   bool first = true;
-
   // Iterate through all members of T using reflection
-  template for (constexpr auto mem : std::define_static_array(
-      std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
-
+  static constexpr auto members = std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()));
+  template for (constexpr auto mem : members) {
     if constexpr (std::meta::is_public(mem)) {
-      constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
+      static constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));

       // Only serialize this field if it's in our list of requested fields
-      if constexpr (should_extract(key)) {
-        if (!first) {
-          b.append(',');
+      if constexpr (((FieldNames.view() == key) || ...)) {
+        static constexpr auto first_key = std::define_static_string(
+            constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)) + ":");
+        static constexpr auto rest_key = std::define_static_string(
+            std::string(",") + constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)) + ":");
+        constexpr size_t first_key_len = std::char_traits<char>::length(first_key);
+        constexpr size_t rest_key_len = std::char_traits<char>::length(rest_key);
+        if (!w.ensure(rest_key_len)) { w.sync(); return; }
+        if (first) {
+          std::memcpy(w.ptr + w.pos, first_key, first_key_len);
+          w.pos += first_key_len;
+        } else {
+          std::memcpy(w.ptr + w.pos, rest_key, rest_key_len);
+          w.pos += rest_key_len;
         }
         first = false;
-
-        // Serialize the key
-        constexpr auto quoted_key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)));
-        b.append_raw(quoted_key);
-        b.append(':');
-
-        // Serialize the value
-        atom(b, obj.[:mem:]);
+        atom(w, obj.[:mem:]);
       }
     }
   };

-  b.append('}');
+  if (!w.ensure(1)) { w.sync(); return; }
+  w.ptr[w.pos++] = '}';
+  w.sync();
 }

 template<constevalutil::fixed_string... FieldNames, typename T>
@@ -44388,25 +53701,18 @@ simdjson_warn_unused simdjson_result<std::string> extract_from(const T &obj, siz
   return std::string(s);
 }

+SIMDJSON_POP_DISABLE_WARNINGS
+
 } // namespace builder
 } // namespace haswell
 // Alias the function template to 'to' in the global namespace
 template <class Z>
 simdjson_warn_unused simdjson_result<std::string> to_json(const Z &z, size_t initial_capacity = haswell::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
-  haswell::builder::string_builder b(initial_capacity);
-  haswell::builder::append(b, z);
-  std::string_view s;
-  if(auto e = b.view().get(s); e) { return e; }
-  return std::string(s);
+  return haswell::builder::to_json_string(z, initial_capacity);
 }
 template <class Z>
 simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = haswell::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
-  haswell::builder::string_builder b(initial_capacity);
-  haswell::builder::append(b, z);
-  std::string_view view;
-  if(auto e = b.view().get(view); e) { return e; }
-  s.assign(view);
-  return SUCCESS;
+  return haswell::builder::to_json(z, s, initial_capacity);
 }
 // Global namespace function for extract_from
 template<constevalutil::fixed_string... FieldNames, typename T>
@@ -44552,6 +53858,7 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
 /* including simdjson/generic/builder/json_string_builder-inl.h for haswell: #include "simdjson/generic/builder/json_string_builder-inl.h" */
 /* begin file simdjson/generic/builder/json_string_builder-inl.h for haswell */
 #include <array>
+#include <cmath>
 #include <cstring>
 #include <limits>
 #include <type_traits>
@@ -44586,6 +53893,11 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
 #define SIMDJSON_EXPERIMENTAL_HAS_LSX 1
 #endif
 #endif
+#if defined(__loongarch_asx)
+#ifndef SIMDJSON_EXPERIMENTAL_HAS_LASX
+#define SIMDJSON_EXPERIMENTAL_HAS_LASX 1
+#endif
+#endif
 #if defined(__riscv_v_intrinsic) && __riscv_v_intrinsic >= 11000 &&            \
     defined(__riscv_vector)
 #ifndef SIMDJSON_EXPERIMENTAL_HAS_RVV
@@ -44605,6 +53917,9 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
 #endif
 #if SIMDJSON_EXPERIMENTAL_HAS_SSE2
 #include <emmintrin.h>
+#if defined(__AVX2__)
+#include <immintrin.h>
+#endif
 #ifdef _MSC_VER
 #include <intrin.h>
 #endif
@@ -44612,6 +53927,9 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
 #if SIMDJSON_EXPERIMENTAL_HAS_LSX
 #include <lsxintrin.h>
 #endif
+#if SIMDJSON_EXPERIMENTAL_HAS_LASX
+#include <lasxintrin.h>
+#endif
 #if SIMDJSON_EXPERIMENTAL_HAS_RVV
 #include <riscv_vector.h>
 #endif
@@ -44661,105 +53979,6 @@ inline bool has_json_escapable_byte(uint64_t x) {

 **/

-SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline bool
-simple_needs_escaping(std::string_view v) {
-  for (char c : v) {
-    // a table lookup is faster than a series of comparisons
-    if (json_quotable_character[static_cast<uint8_t>(c)]) {
-      return true;
-    }
-  }
-  return false;
-}
-
-#if SIMDJSON_EXPERIMENTAL_HAS_NEON
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
-  if (view.size() < 16) {
-    return simple_needs_escaping(view);
-  }
-  size_t i = 0;
-  uint8x16_t running = vdupq_n_u8(0);
-  uint8x16_t v34 = vdupq_n_u8(34);
-  uint8x16_t v92 = vdupq_n_u8(92);
-
-  for (; i + 15 < view.size(); i += 16) {
-    uint8x16_t word = vld1q_u8((const uint8_t *)view.data() + i);
-    running = vorrq_u8(running, vceqq_u8(word, v34));
-    running = vorrq_u8(running, vceqq_u8(word, v92));
-    running = vorrq_u8(running, vcltq_u8(word, vdupq_n_u8(32)));
-  }
-  if (i < view.size()) {
-    uint8x16_t word =
-        vld1q_u8((const uint8_t *)view.data() + view.length() - 16);
-    running = vorrq_u8(running, vceqq_u8(word, v34));
-    running = vorrq_u8(running, vceqq_u8(word, v92));
-    running = vorrq_u8(running, vcltq_u8(word, vdupq_n_u8(32)));
-  }
-  return vmaxvq_u32(vreinterpretq_u32_u8(running)) != 0;
-}
-#elif SIMDJSON_EXPERIMENTAL_HAS_SSE2
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
-  if (view.size() < 16) {
-    return simple_needs_escaping(view);
-  }
-  size_t i = 0;
-  __m128i running = _mm_setzero_si128();
-  for (; i + 15 < view.size(); i += 16) {
-
-    __m128i word =
-        _mm_loadu_si128(reinterpret_cast<const __m128i *>(view.data() + i));
-    running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(34)));
-    running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(92)));
-    running = _mm_or_si128(
-        running, _mm_cmpeq_epi8(_mm_subs_epu8(word, _mm_set1_epi8(31)),
-                                _mm_setzero_si128()));
-  }
-  if (i < view.size()) {
-    __m128i word = _mm_loadu_si128(
-        reinterpret_cast<const __m128i *>(view.data() + view.length() - 16));
-    running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(34)));
-    running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(92)));
-    running = _mm_or_si128(
-        running, _mm_cmpeq_epi8(_mm_subs_epu8(word, _mm_set1_epi8(31)),
-                                _mm_setzero_si128()));
-  }
-  return _mm_movemask_epi8(running) != 0;
-}
-#elif SIMDJSON_EXPERIMENTAL_HAS_PPC64
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
-  if (view.size() < 16) {
-    return simple_needs_escaping(view);
-  }
-  size_t i = 0;
-  __vector unsigned char running = vec_splats((unsigned char)0);
-  __vector unsigned char v34 = vec_splats((unsigned char)34);
-  __vector unsigned char v92 = vec_splats((unsigned char)92);
-  __vector unsigned char v32 = vec_splats((unsigned char)32);
-
-  for (; i + 15 < view.size(); i += 16) {
-    __vector unsigned char word =
-        vec_vsx_ld(0, reinterpret_cast<const unsigned char *>(view.data() + i));
-    running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v34));
-    running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v92));
-    running = vec_or(running,
-        (__vector unsigned char)vec_cmplt(word, v32));
-  }
-  if (i < view.size()) {
-    __vector unsigned char word = vec_vsx_ld(
-        0, reinterpret_cast<const unsigned char *>(view.data() + view.length() - 16));
-    running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v34));
-    running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v92));
-    running = vec_or(running,
-        (__vector unsigned char)vec_cmplt(word, v32));
-  }
-  return !vec_all_eq(running, vec_splats((unsigned char)0));
-}
-#else
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
-  return simple_needs_escaping(view);
-}
-#endif
-
 // Scalar fallback for finding next quotable character
 SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline size_t
 find_next_json_quotable_character_scalar(const std::string_view view,
@@ -44850,6 +54069,51 @@ find_next_json_quotable_character(const std::string_view view,
   size_t current = len - remaining;
   return find_next_json_quotable_character_scalar(view, current);
 }
+#elif SIMDJSON_EXPERIMENTAL_HAS_LASX
+simdjson_inline size_t
+find_next_json_quotable_character(const std::string_view view,
+                                  size_t location) noexcept {
+  const size_t len = view.size();
+  const uint8_t *ptr =
+      reinterpret_cast<const uint8_t *>(view.data()) + location;
+  size_t remaining = len - location;
+
+  // SIMD constants for characters requiring escape
+  __m256i v34 = __lasx_xvreplgr2vr_b(34);  // '"'
+  __m256i v92 = __lasx_xvreplgr2vr_b(92);  // '\\'
+  __m256i v32 = __lasx_xvreplgr2vr_b(32);  // control char threshold
+
+  while (remaining >= 32) {
+    __m256i word = __lasx_xvld(ptr, 0);
+
+    // Check for quotable characters: '"', '\\', or control chars (< 32)
+    __m256i needs_escape = __lasx_xvseq_b(word, v34);
+    needs_escape = __lasx_xvor_v(needs_escape, __lasx_xvseq_b(word, v92));
+    needs_escape = __lasx_xvor_v(needs_escape, __lasx_xvslt_bu(word, v32));
+
+    if (!__lasx_xbz_v(needs_escape)) {
+      // Found a quotable character - locate it via the four 64-bit lanes
+      uint64_t lane0 = __lasx_xvpickve2gr_du(needs_escape, 0);
+      uint64_t lane1 = __lasx_xvpickve2gr_du(needs_escape, 1);
+      uint64_t lane2 = __lasx_xvpickve2gr_du(needs_escape, 2);
+      uint64_t lane3 = __lasx_xvpickve2gr_du(needs_escape, 3);
+      size_t offset = ptr - reinterpret_cast<const uint8_t *>(view.data());
+      if (lane0 != 0) {
+        return offset + trailing_zeroes(lane0) / 8;
+      } else if (lane1 != 0) {
+        return offset + 8 + trailing_zeroes(lane1) / 8;
+      } else if (lane2 != 0) {
+        return offset + 16 + trailing_zeroes(lane2) / 8;
+      } else {
+        return offset + 24 + trailing_zeroes(lane3) / 8;
+      }
+    }
+    ptr += 32;
+    remaining -= 32;
+  }
+  size_t current = len - remaining;
+  return find_next_json_quotable_character_scalar(view, current);
+}
 #elif SIMDJSON_EXPERIMENTAL_HAS_LSX
 simdjson_inline size_t
 find_next_json_quotable_character(const std::string_view view,
@@ -45005,6 +54269,254 @@ SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline void escape_json_char(char c, char *&o
   }
 }

+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2 || SIMDJSON_EXPERIMENTAL_HAS_NEON
+#define SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE 1
+#endif
+
+#if SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2
+
+using escape_vector = __m128i;
+// Mask bits that each input byte contributes to escape_bitmask().
+static constexpr unsigned escape_mask_bits = 1;
+
+simdjson_inline escape_vector escape_load16(const uint8_t *p) noexcept {
+  return _mm_loadu_si128(reinterpret_cast<const __m128i *>(p));
+}
+
+simdjson_inline void escape_store16(char *out, escape_vector v) noexcept {
+  _mm_storeu_si128(reinterpret_cast<__m128i *>(out), v);
+}
+
+// Builds a vector whose bytes 0..7 come from a and bytes 8..15 from b.
+simdjson_inline escape_vector escape_load8x2(const uint8_t *a,
+                                             const uint8_t *b) noexcept {
+  return _mm_unpacklo_epi64(
+      _mm_loadl_epi64(reinterpret_cast<const __m128i *>(a)),
+      _mm_loadl_epi64(reinterpret_cast<const __m128i *>(b)));
+}
+
+// Builds a vector whose bytes 0..3 come from a and bytes 4..7 from b. The
+// upper half is unspecified; callers only look at the low eight bits.
+simdjson_inline escape_vector escape_load4x2(const uint8_t *a,
+                                             const uint8_t *b) noexcept {
+  int32_t a32, b32;
+  memcpy(&a32, a, 4);
+  memcpy(&b32, b, 4);
+  return _mm_unpacklo_epi32(_mm_cvtsi32_si128(a32), _mm_cvtsi32_si128(b32));
+}
+
+// Sets every bit of byte k when byte k of v is a quotable character ('"',
+// '\\' or a control character).
+simdjson_inline escape_vector escape_flags(escape_vector v) noexcept {
+  const __m128i v34 = _mm_set1_epi8(34); // '"'
+  const __m128i v92 = _mm_set1_epi8(92); // '\\'
+  const __m128i v31 = _mm_set1_epi8(31); // for control char detection
+  __m128i needs_escape = _mm_cmpeq_epi8(v, v34);
+  needs_escape = _mm_or_si128(needs_escape, _mm_cmpeq_epi8(v, v92));
+  return _mm_or_si128(
+      needs_escape, _mm_cmpeq_epi8(_mm_subs_epu8(v, v31), _mm_setzero_si128()));
+}
+
+// True when any byte needs escaping. Kept separate from escape_bitmask
+// because some instruction sets can answer it without leaving the vector
+// register file; on SSE2 the compiler folds the two together.
+simdjson_inline bool escape_any(escape_vector flags) noexcept {
+  return _mm_movemask_epi8(flags) != 0;
+}
+
+// A 16-bit mask with bit k set when byte k needed escaping.
+simdjson_inline uint64_t escape_bitmask(escape_vector flags) noexcept {
+  return uint64_t(uint32_t(_mm_movemask_epi8(flags)));
+}
+
+#else // SIMDJSON_EXPERIMENTAL_HAS_NEON
+
+using escape_vector = uint8x16_t;
+static constexpr unsigned escape_mask_bits = 4;
+
+simdjson_inline escape_vector escape_load16(const uint8_t *p) noexcept {
+  return vld1q_u8(p);
+}
+
+simdjson_inline void escape_store16(char *out, escape_vector v) noexcept {
+  vst1q_u8(reinterpret_cast<uint8_t *>(out), v);
+}
+
+simdjson_inline escape_vector escape_load8x2(const uint8_t *a,
+                                             const uint8_t *b) noexcept {
+  uint64_t a64, b64;
+  memcpy(&a64, a, 8);
+  memcpy(&b64, b, 8);
+  return vreinterpretq_u8_u64(vsetq_lane_u64(b64, vdupq_n_u64(a64), 1));
+}
+
+simdjson_inline escape_vector escape_load4x2(const uint8_t *a,
+                                             const uint8_t *b) noexcept {
+  uint32_t a32, b32;
+  memcpy(&a32, a, 4);
+  memcpy(&b32, b, 4);
+  return vreinterpretq_u8_u32(vsetq_lane_u32(b32, vdupq_n_u32(a32), 1));
+}
+
+simdjson_inline escape_vector escape_flags(escape_vector v) noexcept {
+  uint8x16_t needs_escape = vceqq_u8(v, vdupq_n_u8(34));              // '"'
+  needs_escape = vorrq_u8(needs_escape, vceqq_u8(v, vdupq_n_u8(92))); // '\\'
+  return vorrq_u8(needs_escape, vcltq_u8(v, vdupq_n_u8(32)));
+}
+
+simdjson_inline bool escape_any(escape_vector flags) noexcept {
+  return vmaxvq_u32(vreinterpretq_u32_u8(flags)) != 0;
+}
+
+// Four bits per byte rather than one: escape_block and the tail paths scale
+// their shifts by escape_mask_bits to match.
+simdjson_inline uint64_t escape_bitmask(escape_vector flags) noexcept {
+  uint8x8_t narrowed = vshrn_n_u16(vreinterpretq_u16_u8(flags), 4);
+  return vget_lane_u64(vreinterpret_u64_u8(narrowed), 0);
+}
+
+#endif // instruction set selection
+
+// The escape bitmask of a 16-byte block.
+simdjson_inline uint64_t escape_mask(escape_vector v) noexcept {
+  return escape_bitmask(escape_flags(v));
+}
+
+// Copies n bytes with n < 16, using overlapping loads and stores. It never
+// reads more than n bytes from src, nor writes more than n bytes to dst.
+simdjson_inline void copy_lt16(char *dst, const uint8_t *src,
+                               size_t n) noexcept {
+  if (n >= 8) {
+    memcpy(dst, src, 8);
+    memcpy(dst + n - 8, src + n - 8, 8);
+  } else if (n >= 4) {
+    memcpy(dst, src, 4);
+    memcpy(dst + n - 4, src + n - 4, 4);
+  } else if (n > 0) {
+    dst[0] = char(src[0]);
+    dst[n >> 1] = char(src[n >> 1]);
+    dst[n - 1] = char(src[n - 1]);
+  }
+}
+
+// Escapes the bytes of src in the range [i, blockend), given that m is the
+// (non-zero) escape mask for that range: the escape_mask_bits-wide lane of m
+// at byte k is non-zero when src[i + k] requires escaping. Returns the updated
+// output pointer.
+simdjson_never_inline char *escape_block(const uint8_t *src, char *out,
+                                         size_t i, size_t blockend,
+                                         uint64_t m) noexcept {
+  constexpr uint64_t lane = (uint64_t(1) << escape_mask_bits) - 1;
+
+  size_t pos = i; // first byte not yet copied
+  while (m) {
+    const size_t tz = trailing_zeroes(m);
+    const size_t next = i + tz / escape_mask_bits;
+    // Copy the run of safe bytes that precedes this escape.
+    copy_lt16(out, src + pos, next - pos);
+    out += next - pos;
+    escape_json_char(char(src[next]), out);
+    pos = next + 1;
+    m &= ~(lane << tz);
+  }
+  // Copy whatever follows the last escape.
+  copy_lt16(out, src + pos, blockend - pos);
+  return out + (blockend - pos);
+}
+
+// Writes the escaped version of input to out, returning the number of bytes
+// written.
+simdjson_really_inline size_t write_string_escaped(const std::string_view input, char *out) {
+  const size_t len = input.size();
+  const uint8_t *src = reinterpret_cast<const uint8_t *>(input.data());
+  const char *const initout = out;
+
+  size_t i = 0;
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2 && defined(__AVX2__)
+  while (i + 32 <= len) {
+    const __m256i word = _mm256_loadu_si256(reinterpret_cast<const __m256i *>(src + i));
+    const __m256i flags = _mm256_or_si256(
+        _mm256_or_si256(_mm256_cmpeq_epi8(word, _mm256_set1_epi8(34)),   // '"'
+                        _mm256_cmpeq_epi8(word, _mm256_set1_epi8(92))),  // '\\'
+        _mm256_cmpeq_epi8(_mm256_subs_epu8(word, _mm256_set1_epi8(31)),
+                          _mm256_setzero_si256()));                      // control
+    const uint32_t mask = uint32_t(_mm256_movemask_epi8(flags));
+    if (simdjson_likely(mask == 0)) {
+      _mm256_storeu_si256(reinterpret_cast<__m256i *>(out), word);
+      out += 32;
+    } else {
+      for (size_t half = 0; half < 32; half += 16) {
+        const uint64_t m = (mask >> half) & 0xFFFF;
+        if (m == 0) {
+          escape_store16(out, escape_load16(src + i + half));
+          out += 16;
+        } else {
+          out = escape_block(src, out, i + half, i + half + 16, m);
+        }
+      }
+    }
+    i += 32;
+  }
+#endif
+  while (i + 16 <= len) {
+    escape_vector word = escape_load16(src + i);
+    escape_vector flags = escape_flags(word);
+    if (simdjson_likely(!escape_any(flags))) {
+      escape_store16(out, word);
+      out += 16;
+    } else {
+      out = escape_block(src, out, i, i + 16, escape_bitmask(flags));
+    }
+    i += 16;
+  }
+  if (i < len) {
+    const size_t rem = len - i;
+    uint64_t m;
+    if (len >= 16) {
+      // The last 16 bytes of the input are in bounds. Bit k of that block's
+      // mask belongs to input position len - 16 + k, so shift it down to align
+      // bit 0 with position i.
+      m = escape_mask(escape_load16(src + len - 16)) >>
+          (escape_mask_bits * (16 - rem));
+    } else if (len >= 8) {
+      // Here i == 0 and rem == len. Two overlapping 8-byte loads cover
+      // [0, 8) and [len - 8, len), which is the whole input since len < 16.
+      uint64_t mm = escape_mask(escape_load8x2(src, src + len - 8));
+      constexpr uint64_t low8 = (uint64_t(1) << (escape_mask_bits * 8)) - 1;
+      m = (mm & low8) |
+          ((mm >> (escape_mask_bits * 8)) << (escape_mask_bits * (len - 8)));
+    } else if (len >= 4) {
+      // Same idea with two overlapping 4-byte loads.
+      uint64_t mm = escape_mask(escape_load4x2(src, src + len - 4));
+      constexpr uint64_t low4 = (uint64_t(1) << (escape_mask_bits * 4)) - 1;
+      m = (mm & low4) | (((mm >> (escape_mask_bits * 4)) & low4)
+                         << (escape_mask_bits * (len - 4)));
+    } else {
+      // Fewer than 4 bytes: at most three table lookups, no need for SIMD.
+      for (size_t k = 0; k < len; k++) {
+        uint8_t c = src[k];
+        if (json_quotable_character[c]) {
+          escape_json_char(char(c), out);
+        } else {
+          *out++ = char(c);
+        }
+      }
+      return size_t(out - initout);
+    }
+    if (m == 0) {
+      copy_lt16(out, src + i, rem);
+      out += rem;
+    } else {
+      out = escape_block(src, out, i, len, m);
+    }
+  }
+  return size_t(out - initout);
+}
+
+#else // SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
 // Writes the escaped version of input to out, returning the number of bytes
 // written. Uses SIMD position finding to locate quotable characters efficiently.
 inline size_t write_string_escaped(const std::string_view input, char *out) {
@@ -45034,9 +54546,13 @@ inline size_t write_string_escaped(const std::string_view input, char *out) {
     escape_json_char(input[location], out);
     location += 1;
   }
-  return out - initout;
+  return size_t(out - initout);
 }

+#endif // SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+#undef SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
+
 simdjson_inline string_builder::string_builder(size_t initial_capacity)
     : buffer(new(std::nothrow) char[initial_capacity]), position(0),
       capacity(buffer.get() != nullptr ? initial_capacity : 0),
@@ -45059,7 +54575,7 @@ simdjson_inline bool string_builder::capacity_check(size_t upcoming_bytes) {
   return is_valid;
 }

-simdjson_inline void string_builder::grow_buffer(size_t desired_capacity) {
+inline void string_builder::grow_buffer(size_t desired_capacity) {
   if (!is_valid) {
     return;
   }
@@ -45115,81 +54631,136 @@ simdjson_inline void string_builder::clear() noexcept {

 namespace internal {

-template <typename number_type, typename = typename std::enable_if<
-                                    std::is_unsigned<number_type>::value>::type>
-simdjson_really_inline int int_log2(number_type x) {
-  return 63 - leading_zeroes(uint64_t(x) | 1);
-}
-
-simdjson_really_inline int fast_digit_count_32(uint32_t x) {
-  static uint64_t table[] = {
-      4294967296,  8589934582,  8589934582,  8589934582,  12884901788,
-      12884901788, 12884901788, 17179868184, 17179868184, 17179868184,
-      21474826480, 21474826480, 21474826480, 21474826480, 25769703776,
-      25769703776, 25769703776, 30063771072, 30063771072, 30063771072,
-      34349738368, 34349738368, 34349738368, 34349738368, 38554705664,
-      38554705664, 38554705664, 41949672960, 41949672960, 41949672960,
-      42949672960, 42949672960};
-  return uint32_t((x + table[int_log2(x)]) >> 32);
-}
-
-simdjson_really_inline int fast_digit_count_64(uint64_t x) {
-  static uint64_t table[] = {9,
-                             99,
-                             999,
-                             9999,
-                             99999,
-                             999999,
-                             9999999,
-                             99999999,
-                             999999999,
-                             9999999999,
-                             99999999999,
-                             999999999999,
-                             9999999999999,
-                             99999999999999,
-                             999999999999999ULL,
-                             9999999999999999ULL,
-                             99999999999999999ULL,
-                             999999999999999999ULL,
-                             9999999999999999999ULL};
-  int y = (19 * int_log2(x) >> 6);
-  y += x > table[y];
-  return y + 1;
-}
-
-template <typename number_type, typename = typename std::enable_if<
-                                    std::is_unsigned<number_type>::value>::type>
-simdjson_really_inline size_t digit_count(number_type v) noexcept {
-  static_assert(sizeof(number_type) == 8 || sizeof(number_type) == 4 ||
-                    sizeof(number_type) == 2 || sizeof(number_type) == 1,
-                "We only support 8-bit, 16-bit, 32-bit and 64-bit numbers");
-  SIMDJSON_IF_CONSTEXPR(sizeof(number_type) <= 4) {
-    return fast_digit_count_32(static_cast<uint32_t>(v));
+// Integer to decimal: James Edward Anhalt III's algorithm
+static const char jeaiii_dd[201] =
+    "00010203040506070809101112131415161718192021222324252627282930313233343536373839"
+    "40414243444546474849505152535455565758596061626364656667686970717273747576777879"
+    "8081828384858687888990919293949596979899";
+static const char jeaiii_fd[201] =
+    "0\0" "1\0" "2\0" "3\0" "4\0" "5\0" "6\0" "7\0" "8\0" "9\0"
+    "10111213141516171819202122232425262728293031323334353637383940414243444546474849"
+    "50515253545556575859606162636465666768697071727374757677787980818283848586878889"
+    "90919293949596979899";
+
+simdjson_really_inline void jeaiii_write_dd(char *p, uint64_t k) noexcept {
+  std::memcpy(p, &jeaiii_dd[2 * k], 2);
+}
+simdjson_really_inline void jeaiii_write_fd(char *p, uint64_t k) noexcept {
+  std::memcpy(p, &jeaiii_fd[2 * k], 2);
+}
+
+// Caller guarantees n < 10^8. Writes 1 to 8 digits.
+simdjson_really_inline char *jeaiii_lt1e8(char *b, uint32_t n) noexcept {
+  constexpr uint64_t mask24 = (uint64_t(1) << 24) - 1;
+  constexpr uint64_t mask32 = (uint64_t(1) << 32) - 1;
+  if (n < 100) {
+    jeaiii_write_fd(b, n);
+    return n < 10 ? b + 1 : b + 2;
+  }
+  if (n < 1000000) {
+    if (n < 10000) {
+      const uint32_t f0 = uint32_t(10 * (1 << 24) / 1e3 + 1) * n;
+      jeaiii_write_fd(b, f0 >> 24);
+      b -= n < 1000;
+      const uint32_t f2 = uint32_t(f0 & mask24) * 100;
+      jeaiii_write_dd(b + 2, f2 >> 24);
+      return b + 4;
+    }
+    const uint64_t f0 = uint64_t(10 * (1ull << 32) / 1e5 + 1) * n;
+    jeaiii_write_fd(b, f0 >> 32);
+    b -= n < 100000;
+    const uint64_t f2 = (f0 & mask32) * 100;
+    jeaiii_write_dd(b + 2, f2 >> 32);
+    const uint64_t f4 = (f2 & mask32) * 100;
+    jeaiii_write_dd(b + 4, f4 >> 32);
+    return b + 6;
+  }
+  const uint64_t f0 = uint64_t(10 * (1ull << 48) / 1e7 + 1) * n >> 16;
+  jeaiii_write_fd(b, f0 >> 32);
+  b -= n < 10000000;
+  const uint64_t f2 = (f0 & mask32) * 100;
+  jeaiii_write_dd(b + 2, f2 >> 32);
+  const uint64_t f4 = (f2 & mask32) * 100;
+  jeaiii_write_dd(b + 4, f4 >> 32);
+  const uint64_t f6 = (f4 & mask32) * 100;
+  jeaiii_write_dd(b + 6, f6 >> 32);
+  return b + 8;
+}
+
+// Caller guarantees z < 10^8. Always writes exactly 8 digits.
+simdjson_really_inline char *jeaiii_8_digits(char *b, uint32_t z) noexcept {
+  constexpr uint64_t mask32 = (uint64_t(1) << 32) - 1;
+  const uint64_t f0 = (uint64_t((1ull << 48) / 1e6 + 1) * z >> 16) + 1;
+  jeaiii_write_dd(b, f0 >> 32);
+  const uint64_t f2 = (f0 & mask32) * 100;
+  jeaiii_write_dd(b + 2, f2 >> 32);
+  const uint64_t f4 = (f2 & mask32) * 100;
+  jeaiii_write_dd(b + 4, f4 >> 32);
+  const uint64_t f6 = (f4 & mask32) * 100;
+  jeaiii_write_dd(b + 6, f6 >> 32);
+  return b + 8;
+}
+
+// Caller guarantees 10^8 <= n < 2^32. Writes 9 or 10 digits.
+simdjson_really_inline char *jeaiii_9_or_10(char *b, uint64_t n) noexcept {
+  constexpr uint64_t mask57 = (uint64_t(1) << 57) - 1;
+  const uint64_t f0 = uint64_t(10 * (1ull << 57) / 1e9 + 1) * n;
+  jeaiii_write_fd(b, f0 >> 57);
+  b -= n < 1000000000;
+  const uint64_t f2 = (f0 & mask57) * 100;
+  jeaiii_write_dd(b + 2, f2 >> 57);
+  const uint64_t f4 = (f2 & mask57) * 100;
+  jeaiii_write_dd(b + 4, f4 >> 57);
+  const uint64_t f6 = (f4 & mask57) * 100;
+  jeaiii_write_dd(b + 6, f6 >> 57);
+  const uint64_t f8 = (f6 & mask57) * 100;
+  jeaiii_write_dd(b + 8, f8 >> 57);
+  return b + 10;
+}
+
+simdjson_really_inline char *write_uint_jeaiii(char *b, uint64_t n) noexcept {
+  if (n < 100000000) {
+    return jeaiii_lt1e8(b, uint32_t(n));
+  }
+  if (n < (uint64_t(1) << 32)) {
+    return jeaiii_9_or_10(b, n);
+  }
+  // At least 10 digits: the low 8 digits, and 2 to 12 digits above them.
+  const uint32_t z = uint32_t(n % 100000000);
+  uint64_t u = n / 100000000;
+  if (u < 100000000) {
+    // u has 2 to 8 digits (if u < 10, n would be below 2^32).
+    b = jeaiii_lt1e8(b, uint32_t(u));
+  } else if (u < (uint64_t(1) << 32)) {
+    b = jeaiii_9_or_10(b, u);
+  } else {
+    // u has 11 or 12 digits: split off 8 more.
+    const uint32_t y = uint32_t(u % 100000000);
+    u /= 100000000;
+    b = jeaiii_lt1e8(b, uint32_t(u)); // 3 or 4 digits
+    b = jeaiii_8_digits(b, y);
   }
-  else {
-    return fast_digit_count_64(static_cast<uint64_t>(v));
-  }
-}
-static const char decimal_table[200] = {
-    0x30, 0x30, 0x30, 0x31, 0x30, 0x32, 0x30, 0x33, 0x30, 0x34, 0x30, 0x35,
-    0x30, 0x36, 0x30, 0x37, 0x30, 0x38, 0x30, 0x39, 0x31, 0x30, 0x31, 0x31,
-    0x31, 0x32, 0x31, 0x33, 0x31, 0x34, 0x31, 0x35, 0x31, 0x36, 0x31, 0x37,
-    0x31, 0x38, 0x31, 0x39, 0x32, 0x30, 0x32, 0x31, 0x32, 0x32, 0x32, 0x33,
-    0x32, 0x34, 0x32, 0x35, 0x32, 0x36, 0x32, 0x37, 0x32, 0x38, 0x32, 0x39,
-    0x33, 0x30, 0x33, 0x31, 0x33, 0x32, 0x33, 0x33, 0x33, 0x34, 0x33, 0x35,
-    0x33, 0x36, 0x33, 0x37, 0x33, 0x38, 0x33, 0x39, 0x34, 0x30, 0x34, 0x31,
-    0x34, 0x32, 0x34, 0x33, 0x34, 0x34, 0x34, 0x35, 0x34, 0x36, 0x34, 0x37,
-    0x34, 0x38, 0x34, 0x39, 0x35, 0x30, 0x35, 0x31, 0x35, 0x32, 0x35, 0x33,
-    0x35, 0x34, 0x35, 0x35, 0x35, 0x36, 0x35, 0x37, 0x35, 0x38, 0x35, 0x39,
-    0x36, 0x30, 0x36, 0x31, 0x36, 0x32, 0x36, 0x33, 0x36, 0x34, 0x36, 0x35,
-    0x36, 0x36, 0x36, 0x37, 0x36, 0x38, 0x36, 0x39, 0x37, 0x30, 0x37, 0x31,
-    0x37, 0x32, 0x37, 0x33, 0x37, 0x34, 0x37, 0x35, 0x37, 0x36, 0x37, 0x37,
-    0x37, 0x38, 0x37, 0x39, 0x38, 0x30, 0x38, 0x31, 0x38, 0x32, 0x38, 0x33,
-    0x38, 0x34, 0x38, 0x35, 0x38, 0x36, 0x38, 0x37, 0x38, 0x38, 0x38, 0x39,
-    0x39, 0x30, 0x39, 0x31, 0x39, 0x32, 0x39, 0x33, 0x39, 0x34, 0x39, 0x35,
-    0x39, 0x36, 0x39, 0x37, 0x39, 0x38, 0x39, 0x39,
-};
+  return jeaiii_8_digits(b, z);
+}
+
+// Writes v at p, which must have to_chars_buffer_size bytes available, and
+// returns the end of what was written.
+simdjson_inline char *write_double(char *p, double v) noexcept {
+#if SIMDJSON_ENABLE_NAN_INF
+  if (simdjson_unlikely(!std::isfinite(v))) {
+    if (std::isnan(v)) {
+      std::memcpy(p, "NaN", 3);
+      return p + 3;
+    }
+    if (v < 0) {
+      *p++ = '-';
+    }
+    std::memcpy(p, "Infinity", 8);
+    return p + 8;
+  }
+#endif
+  return simdjson::internal::to_chars(p, nullptr, v);
+}
 } // namespace internal

 template <typename number_type, typename>
@@ -45217,87 +54788,62 @@ simdjson_inline void string_builder::append(number_type v) noexcept {
     }
   }
   else SIMDJSON_IF_CONSTEXPR(std::is_unsigned<number_type>::value) {
-    // Process 4 digits at a time instead of 2, reducing store operations
-    // and divisions by approximately half for large numbers.
     constexpr size_t max_number_size = 20;
     if (capacity_check(max_number_size)) {
       using unsigned_type = typename std::make_unsigned<number_type>::type;
-      unsigned_type pv = static_cast<unsigned_type>(v);
-      size_t dc = internal::digit_count(pv);
-      char *write_pointer = buffer.get() + position + dc - 1;
-
-      // Process 4 digits per iteration for large numbers
-      while (pv >= 10000) {
-        unsigned_type q = pv / 10000;
-        unsigned_type r = pv % 10000;
-        unsigned_type r_hi = r / 100;  // High 2 digits of remainder
-        unsigned_type r_lo = r % 100;  // Low 2 digits of remainder
-        // Write low 2 digits first (rightmost), then high 2 digits
-        memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
-        memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
-        write_pointer -= 4;
-        pv = q;
-      }
-
-      // Handle remaining 1-4 digits with original 2-digit loop
-      while (pv >= 100) {
-        memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
-        write_pointer -= 2;
-        pv /= 100;
-      }
-      if (pv >= 10) {
-        *write_pointer-- = char('0' + (pv % 10));
-        pv /= 10;
-      }
-      *write_pointer = char('0' + pv);
-      position += dc;
+      char* end = internal::write_uint_jeaiii(
+          buffer.get() + position,
+          static_cast<uint64_t>(static_cast<unsigned_type>(v)));
+      position = end - buffer.get();
     }
   }
   else SIMDJSON_IF_CONSTEXPR(std::is_integral<number_type>::value) {
-    // Same 4-digit batching as unsigned path for signed integers
+    // 19 digits (max abs value of int64_t) + optional minus sign.
     constexpr size_t max_number_size = 20;
     if (capacity_check(max_number_size)) {
       using unsigned_type = typename std::make_unsigned<number_type>::type;
       bool negative = v < 0;
-      unsigned_type pv = static_cast<unsigned_type>(v);
-      if (negative) {
-        pv = 0 - pv; // the 0 is for Microsoft
-      }
-      size_t dc = internal::digit_count(pv);
-      // by always writing the minus sign, we avoid the branch.
+      // 0 - pv (rather than -pv) avoids an MSVC unary-minus warning.
+      unsigned_type pv = negative
+          ? unsigned_type(0) - static_cast<unsigned_type>(v)
+          : static_cast<unsigned_type>(v);
+      // Branchless: always write '-', advance only if negative.
       buffer.get()[position] = '-';
-      position += negative ? 1 : 0;
-      char *write_pointer = buffer.get() + position + dc - 1;
-
-      // Process 4 digits per iteration for large numbers
-      while (pv >= 10000) {
-        unsigned_type q = pv / 10000;
-        unsigned_type r = pv % 10000;
-        unsigned_type r_hi = r / 100;
-        unsigned_type r_lo = r % 100;
-        memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
-        memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
-        write_pointer -= 4;
-        pv = q;
-      }
-
-      // Handle remaining 1-4 digits
-      while (pv >= 100) {
-        memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
-        write_pointer -= 2;
-        pv /= 100;
-      }
-      if (pv >= 10) {
-        *write_pointer-- = char('0' + (pv % 10));
-        pv /= 10;
-      }
-      *write_pointer = char('0' + pv);
-      position += dc;
+      position += negative;
+      char* end = internal::write_uint_jeaiii(
+          buffer.get() + position, static_cast<uint64_t>(pv));
+      position = end - buffer.get();
     }
   }
   else SIMDJSON_IF_CONSTEXPR(std::is_floating_point<number_type>::value) {
-    constexpr size_t max_number_size = 24;
+    // Must reserve to_chars_buffer_size (40): only ~24 chars are emitted,
+    // but to_chars over-writes with fixed-size 16/17-byte copies so the
+    // compiler can inline mem* (see simdjson::internal::to_chars_buffer_size).
+    constexpr size_t max_number_size = simdjson::internal::to_chars_buffer_size;
     if (capacity_check(max_number_size)) {
+#if SIMDJSON_ENABLE_NAN_INF
+      // Check if the input might be NaN or infinity
+      if (simdjson_unlikely(!std::isfinite(v))) {
+        if (std::isnan(v)) {
+          constexpr char nan_literal[] = "NaN";
+          constexpr size_t nan_len = sizeof(nan_literal) - 1;
+
+          std::memcpy(buffer.get() + position, nan_literal, nan_len);
+          position += nan_len;
+        } else {
+          constexpr char inf_literal[] = "Infinity";
+          constexpr size_t inf_len = sizeof(inf_literal) - 1;
+          if (v < 0) {
+            buffer.get()[position] = '-';
+            ++position;
+          }
+          std::memcpy(buffer.get() + position, inf_literal, inf_len);
+          position += inf_len;
+        }
+        return;
+      }
+#endif
+
       // We could specialize for float.
       char *end = simdjson::internal::to_chars(buffer.get() + position, nullptr,
                                                double(v));
@@ -45358,7 +54904,9 @@ simdjson_inline void string_builder::escape_and_append_with_quotes() noexcept {
 #endif

 simdjson_inline void string_builder::append_raw(const char *c) noexcept {
-  size_t len = std::strlen(c);
+  // char_traits::length is constexpr; lets the compiler fold the length
+  // when called with a pointer to a compile-time-constant string.
+  size_t len = std::char_traits<char>::length(c);
   append_raw(c, len);
 }

@@ -45377,6 +54925,14 @@ simdjson_inline void string_builder::append_raw(const char *str,
     position += len;
   }
 }
+
+template <size_t N>
+simdjson_inline void string_builder::append_raw_n(const char *str) noexcept {
+  if (capacity_check(N)) {
+    std::memcpy(buffer.get() + position, str, N);
+    position += N;
+  }
+}
 #if SIMDJSON_SUPPORTS_CONCEPTS
 // Support for optional types (std::optional, etc.)
 template <concepts::optional_type T>
@@ -45406,7 +54962,7 @@ simdjson_inline void string_builder::append(const T &value) {
 #if SIMDJSON_SUPPORTS_RANGES && SIMDJSON_SUPPORTS_CONCEPTS
 // Support for range-based appending (std::ranges::view, etc.)
 template <std::ranges::range R>
-  requires(!std::is_convertible<R, std::string_view>::value && !require_custom_serialization<R>)
+  requires(!std::is_convertible<R, std::string_view>::value && !concepts::optional_type<R> && !require_custom_serialization<R>)
 simdjson_inline void string_builder::append(const R &range) noexcept {
   auto it = std::ranges::begin(range);
   auto end = std::ranges::end(range);
@@ -45748,16 +55304,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
 }
 #endif

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
-                                uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
-  return _addcarry_u64(0, value1, value2,
-                       reinterpret_cast<unsigned __int64 *>(result));
-#else
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-#endif
-}

 } // unnamed namespace
 } // namespace icelake
@@ -46499,7 +56045,7 @@ public:
 #if SIMDJSON_SUPPORTS_RANGES && SIMDJSON_SUPPORTS_CONCEPTS
   // Support for range-based appending (std::ranges::view, etc.)
   template <std::ranges::range R>
-requires (!std::is_convertible<R, std::string_view>::value && !require_custom_serialization<R>)
+requires (!std::is_convertible<R, std::string_view>::value && !concepts::optional_type<R> && !require_custom_serialization<R>)
   simdjson_inline void append(const R &range) noexcept;
 #endif
   /**
@@ -46513,6 +56059,15 @@ requires (!std::is_convertible<R, std::string_view>::value && !require_custom_se
    * There is no UTF-8 validation.
    */
   simdjson_inline void append_raw(const char *str, size_t len) noexcept;
+
+  /**
+   * Append exactly N characters from str. The length is a template parameter
+   * so the compiler can fully inline the memcpy with a compile-time-constant
+   * size, avoiding the libc call. Used for compile-time-constant keys in the
+   * reflection struct atom.
+   */
+  template <size_t N>
+  simdjson_inline void append_raw_n(const char *str) noexcept;
 #if SIMDJSON_EXCEPTIONS
   /**
    * Creates an std::string from the written JSON buffer.
@@ -46564,6 +56119,26 @@ requires (!std::is_convertible<R, std::string_view>::value && !require_custom_se
    */
   simdjson_inline size_t size() const noexcept;

+  // ============================================================
+  // Internal hooks for the position-as-local writer in json_builder.h.
+  // These exist so the reflection atom code can hold buffer pointer,
+  // position and capacity in registers across long write chains rather
+  // than reloading them after every char* write (strict aliasing
+  // forces those reloads when accessed via members of *this). User
+  // code should NOT call these directly.
+  // ============================================================
+  simdjson_inline char *unsafe_data() noexcept { return buffer.get(); }
+  simdjson_inline size_t unsafe_position() const noexcept { return position; }
+  simdjson_inline size_t unsafe_capacity() const noexcept { return capacity; }
+  simdjson_inline void unsafe_set_position(size_t p) noexcept { position = p; }
+  /// Make capacity available for at least `n` more bytes after the current
+  /// position. Returns false if the allocation failed.
+  simdjson_inline bool unsafe_grow(size_t needed_total_capacity) noexcept {
+    grow_buffer(needed_total_capacity);
+    return is_valid;
+  }
+  simdjson_inline bool unsafe_is_valid() const noexcept { return is_valid; }
+
 private:
   /**
    * Returns true if we can write at least upcoming_bytes bytes.
@@ -46577,7 +56152,7 @@ private:
    * If the allocation fails, is_valid is set to false. We expect
    * that this function would not be repeatedly called.
    */
-  simdjson_inline void grow_buffer(size_t desired_capacity);
+  inline void grow_buffer(size_t desired_capacity);

   /**
    * We use this helper function to make sure that is_valid is kept consistent.
@@ -46634,6 +56209,7 @@ simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initi
 /* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_STRING_BUILDER_H */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/builder/json_string_builder.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/concepts.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/annotations.h" */
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
 #if SIMDJSON_STATIC_REFLECTION

@@ -46651,64 +56227,370 @@ namespace simdjson {
 namespace icelake {
 namespace builder {

-template <class T>
+// Forward-declare helpers defined in json_string_builder-inl.h so the
+// writer-based atom code below can call them (the -inl.h is not yet
+// included at the point this header is parsed; without these forwards,
+// name lookup falls back to the wrong outer namespace).
+namespace internal {
+simdjson_really_inline char *write_uint_jeaiii(char *p, uint64_t v) noexcept;
+simdjson_inline char *write_double(char *p, double v) noexcept;
+} // namespace internal
+simdjson_really_inline size_t write_string_escaped(const std::string_view input, char *out);
+
+SIMDJSON_PUSH_DISABLE_WARNINGS
+SIMDJSON_DISABLE_GCC_WARNING(-Warray-bounds)
+#if !defined(__clang__)
+SIMDJSON_DISABLE_GCC_WARNING(-Wstringop-overflow)
+#endif
+
+// =============================================================
+// `writer`: position-as-local hot-path writer used by the reflection
+// atom code below. Holds the buffer pointer, write position and
+// capacity in three fields that, once `writer` itself is a stack-local
+// in the caller and all atom() functions are inlined, become true
+// register-resident locals after SROA. Glaze achieves the same effect
+// by passing `B&& b, auto&& ix` through every helper. Holding `pos`
+// in a register (rather than as a member of string_builder) is what
+// breaks the strict-aliasing penalty on every char* write through the
+// buffer, which forces a reload of `b.position` and `b.capacity`
+// after every byte.
+//
+// basic_writer<false> (below) is the unchecked variant: the caller has
+// already reserved enough capacity for everything the write chain can
+// produce (see bound_detail::size_bound), so ensure() compiles away.
+// =============================================================
+template <bool Checked>
+struct basic_writer {
+  static constexpr bool checked = Checked;
+  char *ptr;        // buffer pointer (refreshed after a grow)
+  size_t pos;       // write position (local)
+  size_t cap;       // capacity (refreshed after a grow)
+  string_builder &sb;  // back-ref for grow / sync
+
+  // Snapshot string_builder state into a writer for the duration of
+  // a write chain.
+  simdjson_really_inline basic_writer(string_builder &builder) noexcept
+      : ptr(builder.unsafe_data())
+      , pos(builder.unsafe_position())
+      , cap(builder.unsafe_capacity())
+      , sb(builder) {}
+
+  // Write the local position back to the underlying string_builder.
+  // Caller is responsible for invoking before the writer is dropped
+  // (otherwise data is lost). Idempotent.
+  simdjson_really_inline void sync() noexcept {
+    sb.unsafe_set_position(pos);
+  }
+
+  // Ensure at least `n` more bytes of free capacity. Grows the
+  // underlying buffer if needed (rare path). Returns false on
+  // allocation failure.
+  simdjson_really_inline bool ensure(size_t n) noexcept {
+    // pos <= cap, and cap is the size of a live allocation, so pos + n
+    // cannot wrap when n is a small constant or a compile-time length.
+    // Callers passing a size derived from input (the string atoms) must
+    // bound it against pos themselves. Keep the `pos + n <= cap` form:
+    // `n <= cap - pos` is measurably slower once the serializer is inlined.
+    if (simdjson_likely(pos + n <= cap)) { return true; }
+    return grow_slow(n);
+  }
+
+  simdjson_never_inline bool grow_slow(size_t n) noexcept {
+    // Detect overflow.
+    // This is pedantic except maybe on 32-bit targets.
+    if (simdjson_unlikely(pos + n < pos)) return false;
+    sb.unsafe_set_position(pos);
+    // even if 2*capacity overflows, the (std::max) below will pick the needed value,
+    // so we do not need a separate overflow check here.
+    if (!sb.unsafe_grow((std::max)(cap * 2, pos + n))) {
+      // The string_builder freed its buffer and is now invalid (null buffer,
+      // zero capacity and position). Mirror that state so that every later
+      // ensure() fails too: callers only return from the current atom, and
+      // their callers keep writing.
+      ptr = nullptr;
+      pos = 0;
+      cap = 0;
+      return false;
+    }
+    ptr = sb.unsafe_data();
+    cap = sb.unsafe_capacity();
+    return true;
+  }
+};
+
+// The unchecked writer writes into a raw buffer that the caller sized with
+// serialized_size_bound: it never grows and needs no string_builder.
+template <>
+struct basic_writer<false> {
+  static constexpr bool checked = false;
+  char *ptr;
+  size_t pos;
+
+  simdjson_really_inline basic_writer(char *buffer, size_t position) noexcept
+      : ptr(buffer), pos(position) {}
+
+  simdjson_really_inline bool ensure(size_t) const noexcept { return true; }
+};
+
+using writer = basic_writer<true>;
+using unchecked_writer = basic_writer<false>;
+
+// Bytes reserved past the size bound for an unchecked writer: it may then
+// write a little past the end of what it produces (e.g., copy keys as whole
+// 16-byte blocks).
+inline constexpr size_t unchecked_slack = 64;
+
+consteval size_t padded_key_length(size_t length) {
+  return (length + 15) / 16 * 16;
+}
+
+// === Helper: invoke a string_builder member that writes variable-length
+// content (escape_and_append_with_quotes etc), syncing the writer's local
+// state before the call and reloading after. Used for string fields where
+// rewriting the entire SIMD escape path through the writer would be a much
+// bigger refactor. f may be user code (a with<Adapter> serializer) that
+// throws: the exception then propagates to the caller.
+template <class W, class F>
+simdjson_really_inline void call_through_string_builder(W &w, F &&f) noexcept(noexcept(f(w.sb))) {
+  w.sync();
+  f(w.sb);
+  w.ptr = w.sb.unsafe_data();
+  w.pos = w.sb.unsafe_position();
+  w.cap = w.sb.unsafe_capacity();
+}
+
+// Helpers implementing the serialization side of the annotations (see
+// simdjson/annotations.h) for reflected structures.
+namespace annotation_detail {
+
+// A member is serialized unless it is annotated with skip or skip_serializing.
+consteval bool is_serialized_member(std::meta::info dm) {
+  return !simdjson::detail::has_annotation(dm, ^^simdjson::detail::skip_tag)
+      && !simdjson::detail::has_annotation(dm, ^^simdjson::detail::skip_serializing_tag);
+}
+
+// False when the member has a skip_serializing_if<pred> annotation and
+// pred(value) is true.
+template <auto dm, typename V>
+simdjson_really_inline bool should_serialize(const V &value) {
+  constexpr std::meta::info skip_if_type = simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::skip_serializing_if_t);
+  if constexpr (skip_if_type != std::meta::info{}) {
+    using skip_if = typename [: skip_if_type :];
+    return !skip_if::predicate(value);
+  } else {
+    (void)value;
+    return true;
+  }
+}
+
+// Serialize a member value, through its with<Adapter> annotation when the
+// adapter provides a serialize function.
+template <auto dm, class W, typename V>
+simdjson_really_inline void atom_member(W &w, const V &value) {
+  constexpr std::meta::info with_type = simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::with_t);
+  if constexpr (with_type != std::meta::info{}) {
+    using adapter = typename [: with_type :]::adapter;
+    if constexpr (requires(string_builder &b) { adapter::serialize(b, value); }) {
+      call_through_string_builder(w, [&](string_builder &b) { adapter::serialize(b, value); });
+    } else {
+      atom(w, value);
+    }
+  } else {
+    atom(w, value);
+  }
+}
+
+// Write the "key":value pairs of the members of t (without the braces), each
+// preceded by a comma unless it is the first one. The members of a member
+// annotated with flatten are written in its place.
+template <class W, class T>
+simdjson_really_inline void atom_fields(W &w, const T &t, bool &first) {
+  // Per-field block: ensure key+value worst case, then write key + value
+  // through the writer's local pos. For arithmetic fields, the integer
+  // write happens directly via write_uint_jeaiii on w.ptr+w.pos, so pos
+  // never round-trips through memory.
+  template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+    if constexpr (is_serialized_member(dm)) {
+      if (should_serialize<dm>(t.[:dm:])) {
+        if constexpr (simdjson::detail::has_annotation(dm, ^^simdjson::detail::flatten_tag)) {
+          static_assert(std::meta::is_class_type(simdjson::detail::flattened_type(dm)));
+          using flattened = std::remove_cvref_t<decltype(t.[:dm:])>;
+          static_assert(!concepts::container_but_not_string<flattened> && !concepts::string_view_keyed_map<flattened> &&
+                        !concepts::appendable_containers<flattened> && !concepts::optional_type<flattened> &&
+                        !concepts::smart_pointer<flattened> && !std::is_same_v<flattened, std::string> &&
+                        !std::is_same_v<flattened, std::string_view> && !require_custom_serialization<flattened>,
+                        "simdjson::flatten requires a member whose type is a structure serialized member by member");
+          atom_fields(w, t.[:dm:], first);
+        } else {
+          // Copy the key as whole 16-byte blocks from a zero-padded copy (one
+          // load and one store); ensure() reserves the padded length, and the
+          // unchecked writer has slack past its bound. Prior related work:
+          // jsonifier copies a power-of-two padded key and advances the cursor
+          // by the real length (serialize_impl.hpp, packed_blitter,
+          // https://github.com/nihilai-collective/Jsonifier).
+          constexpr const char* key_name = simdjson::get_json_key_name<dm>();
+          constexpr size_t first_key_len = constevalutil::consteval_to_quoted_escaped(key_name).size() + 1;
+          constexpr size_t rest_key_len = first_key_len + 1;
+          constexpr auto first_key = std::define_static_string(
+              constevalutil::consteval_to_quoted_escaped(key_name) + ":" +
+              std::string(padded_key_length(first_key_len) - first_key_len, '\0'));
+          constexpr auto rest_key = std::define_static_string(
+              std::string(",") + constevalutil::consteval_to_quoted_escaped(key_name) + ":" +
+              std::string(padded_key_length(rest_key_len) - rest_key_len, '\0'));
+          if (!w.ensure(padded_key_length(rest_key_len))) { return; }
+          if (first) {
+            std::memcpy(w.ptr + w.pos, first_key, padded_key_length(first_key_len));
+            w.pos += first_key_len;
+          } else {
+            std::memcpy(w.ptr + w.pos, rest_key, padded_key_length(rest_key_len));
+            w.pos += rest_key_len;
+          }
+          first = false;
+          atom_member<dm>(w, t.[:dm:]);
+        }
+      }
+    }
+  };
+}
+
+} // namespace annotation_detail
+
+template <class W, class T>
   requires(concepts::container_but_not_string<T> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &t) {
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
   auto it = t.begin();
   auto end = t.end();
   if (it == end) {
-    b.append_raw("[]");
+    if (!w.ensure(2)) return;
+    std::memcpy(w.ptr + w.pos, "[]", 2);
+    w.pos += 2;
     return;
   }
-  b.append('[');
-  atom(b, *it);
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = '[';
+  atom(w, *it);
   ++it;
   for (; it != end; ++it) {
-    b.append(',');
-    atom(b, *it);
+    if (!w.ensure(1)) return;
+    w.ptr[w.pos++] = ',';
+    atom(w, *it);
   }
-  b.append(']');
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = ']';
 }

-template <class T>
+template <class W, class T>
   requires(std::is_same_v<T, std::string> ||
            std::is_same_v<T, std::string_view> ||
            std::is_same_v<T, const char *> ||
            std::is_same_v<T, char>)
-constexpr void atom(string_builder &b, const T &t) {
-  b.escape_and_append_with_quotes(t);
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
+  // Inline the escape path through the writer so we never round-trip
+  // pos through memory for string fields (Twitter is dominated by
+  // these -- sync/reload around each string was a real cost).
+  std::string_view input;
+  if constexpr (std::is_same_v<T, char>) {
+    input = std::string_view(&t, 1);
+  } else {
+    input = std::string_view(t);
+  }
+  // Worst-case escape: every byte expands to \uXXXX (6 chars), plus 2 quotes.
+  // Guard against w.pos + 2 + 6 * input.size() wrapping for huge inputs -- if
+  // it wrapped to a small value, ensure() would spuriously succeed and the
+  // subsequent escape would overflow the buffer. max - w.pos cannot wrap, and
+  // size < (max - pos) / 6 implies pos + 6 * size + 6 <= max.
+  // Note that this is pedantic except maybe on 32-bit targets.
+  if constexpr (W::checked) {
+    if (simdjson_unlikely(input.size() >= ((std::numeric_limits<size_t>::max)() - w.pos) / 6)) { return; }
+    if (!w.ensure(2 + 6 * input.size())) { return; }
+  }
+  w.ptr[w.pos++] = '"';
+  w.pos += write_string_escaped(input, w.ptr + w.pos);
+  w.ptr[w.pos++] = '"';
 }

-template <concepts::string_view_keyed_map T>
+template <class W, concepts::string_view_keyed_map T>
   requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &m) {
+simdjson_really_inline constexpr void atom(W &w, const T &m) {
   if (m.empty()) {
-    b.append_raw("{}");
+    if (!w.ensure(2)) return;
+    std::memcpy(w.ptr + w.pos, "{}", 2);
+    w.pos += 2;
     return;
   }
-  b.append('{');
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = '{';
   bool first = true;
   for (const auto& [key, value] : m) {
     if (!first) {
-      b.append(',');
+      if (!w.ensure(1)) return;
+      w.ptr[w.pos++] = ',';
     }
     first = false;
-    // Keys must be convertible to string_view per the concept
-    b.escape_and_append_with_quotes(key);
-    b.append(':');
-    atom(b, value);
+    // Keys must be convertible to string_view per the concept.
+    std::string_view key_sv(key);
+    // Guard against w.pos + 3 + 6 * key_sv.size() wrapping for huge keys, if
+    // it wrapped to a small value, ensure() would spuriously succeed and the
+    // subsequent escape would overflow the buffer. max - w.pos cannot wrap, and
+    // size < (max - pos) / 6 implies pos + 6 * size + 6 <= max.
+    // Note that this is pedantic except maybe on 32-bit targets.
+    if constexpr (W::checked) {
+      if (simdjson_unlikely(key_sv.size() >= ((std::numeric_limits<size_t>::max)() - w.pos) / 6)) { return; }
+      if (!w.ensure(2 + 6 * key_sv.size() + 1)) { return; }
+    }
+    w.ptr[w.pos++] = '"';
+    w.pos += write_string_escaped(key_sv, w.ptr + w.pos);
+    w.ptr[w.pos++] = '"';
+    w.ptr[w.pos++] = ':';
+    atom(w, value);
   }
-  b.append('}');
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = '}';
 }


-template<typename number_type,
+template<class W, typename number_type,
          typename = typename std::enable_if<std::is_arithmetic<number_type>::value && !std::is_same_v<number_type, char>>::type>
-constexpr void atom(string_builder &b, const number_type t) {
-  b.append(t);
+simdjson_really_inline constexpr void atom(W &w, const number_type t) {
+  // Booleans / floats: defer to string_builder (rare path; keeps writer hot
+  // path free of float-formatter machinery). For integers, write directly
+  // via jeaiii using local pos.
+  if constexpr (std::is_same_v<number_type, bool>) {
+    if (t) {
+      if (!w.ensure(4)) return;
+      std::memcpy(w.ptr + w.pos, "true", 4);
+      w.pos += 4;
+    } else {
+      if (!w.ensure(5)) return;
+      std::memcpy(w.ptr + w.pos, "false", 5);
+      w.pos += 5;
+    }
+  } else if constexpr (std::is_floating_point_v<number_type>) {
+    if constexpr (W::checked) {
+      call_through_string_builder(w, [&](string_builder &b) { b.append(t); });
+    } else {
+      w.pos = size_t(internal::write_double(w.ptr + w.pos, double(t)) - w.ptr);
+    }
+  } else if constexpr (std::is_unsigned_v<number_type>) {
+    if (!w.ensure(20)) return;
+    char *end = internal::write_uint_jeaiii(
+        w.ptr + w.pos, static_cast<uint64_t>(t));
+    w.pos = static_cast<size_t>(end - w.ptr);
+  } else {
+    // signed integral
+    if (!w.ensure(20)) return;
+    using U = typename std::make_unsigned<number_type>::type;
+    bool negative = t < 0;
+    U pv = negative ? U(0) - static_cast<U>(t) : static_cast<U>(t);
+    w.ptr[w.pos] = '-';
+    w.pos += negative;
+    char *end = internal::write_uint_jeaiii(
+        w.ptr + w.pos, static_cast<uint64_t>(pv));
+    w.pos = static_cast<size_t>(end - w.ptr);
+  }
 }

-template <class T>
+template <class W, class T>
   requires(std::is_class_v<T> && !concepts::container_but_not_string<T> &&
            !concepts::string_view_keyed_map<T> &&
            !concepts::optional_type<T> &&
@@ -46718,92 +56600,259 @@ template <class T>
            !std::is_same_v<T, std::string_view> &&
            !std::is_same_v<T, const char*> &&
            !std::is_same_v<T, char> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &t) {
-  int i = 0;
-  b.append('{');
-  template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
-    if (i != 0)
-      b.append(',');
-    constexpr auto key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(dm)));
-    b.append_raw(key);
-    b.append(':');
-    atom(b, t.[:dm:]);
-    i++;
-  };
-  b.append('}');
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
+  if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag)) {
+    // A transparent structure is serialized as its single member.
+    constexpr auto dm = simdjson::detail::transparent_member(^^T);
+    annotation_detail::atom_member<dm>(w, t.[:dm:]);
+  } else {
+    bool first = true;
+    if (!w.ensure(1)) { return; }
+    w.ptr[w.pos++] = '{';
+    annotation_detail::atom_fields(w, t, first);
+    if (!w.ensure(1)) { return; }
+    w.ptr[w.pos++] = '}';
+  }
 }

 // Support for optional types (std::optional, etc.)
-template <concepts::optional_type T>
+template <class W, concepts::optional_type T>
   requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &opt) {
+simdjson_really_inline constexpr void atom(W &w, const T &opt) {
   if (opt) {
-    atom(b, opt.value());
+    atom(w, opt.value());
   } else {
-    b.append_raw("null");
+    if (!w.ensure(4)) return;
+    std::memcpy(w.ptr + w.pos, "null", 4);
+    w.pos += 4;
   }
 }

 // Support for smart pointers (std::unique_ptr, std::shared_ptr, etc.)
-template <concepts::smart_pointer T>
+template <class W, concepts::smart_pointer T>
   requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &ptr) {
+simdjson_really_inline constexpr void atom(W &w, const T &ptr) {
   if (ptr) {
-    atom(b, *ptr);
+    atom(w, *ptr);
   } else {
-    b.append_raw("null");
+    if (!w.ensure(4)) return;
+    std::memcpy(w.ptr + w.pos, "null", 4);
+    w.pos += 4;
   }
 }

 // Support for enums - serialize as string representation using expand approach from P2996R12
-template <typename T>
+template <class W, typename T>
   requires(std::is_enum_v<T> && !require_custom_serialization<T>)
-void atom(string_builder &b, const T &e) {
+simdjson_really_inline void atom(W &w, const T &e) {
 #if SIMDJSON_STATIC_REFLECTION
-  constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
+  static constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
   template for (constexpr auto enum_val : enumerators) {
-    constexpr auto enum_str = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(enum_val)));
+    constexpr auto enum_str = std::define_static_string(constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<enum_val>()));
+    constexpr size_t enum_str_len = std::char_traits<char>::length(enum_str);
     if (e == [:enum_val:]) {
-      b.append_raw(enum_str);
+      if (!w.ensure(enum_str_len)) return;
+      std::memcpy(w.ptr + w.pos, enum_str, enum_str_len);
+      w.pos += enum_str_len;
       return;
     }
   };
   // Fallback to integer if enum value not found
-  atom(b, static_cast<std::underlying_type_t<T>>(e));
+  atom(w, static_cast<std::underlying_type_t<T>>(e));
 #else
   // Fallback: serialize as integer if reflection not available
-  atom(b, static_cast<std::underlying_type_t<T>>(e));
+  atom(w, static_cast<std::underlying_type_t<T>>(e));
 #endif
 }

 // Support for appendable containers that don't have operator[] (sets, etc.)
-template <concepts::appendable_containers T>
+template <class W, concepts::appendable_containers T>
   requires(!concepts::container_but_not_string<T> && !concepts::string_view_keyed_map<T> &&
            !concepts::optional_type<T> && !concepts::smart_pointer<T> &&
            !std::is_same_v<T, std::string> &&
            !std::is_same_v<T, std::string_view> && !std::is_same_v<T, const char*> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &container) {
+simdjson_really_inline constexpr void atom(W &w, const T &container) {
   if (container.empty()) {
-    b.append_raw("[]");
+    if (!w.ensure(2)) return;
+    std::memcpy(w.ptr + w.pos, "[]", 2);
+    w.pos += 2;
     return;
   }
-  b.append('[');
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = '[';
   bool first = true;
   for (const auto& item : container) {
     if (!first) {
-      b.append(',');
+      if (!w.ensure(1)) return;
+      w.ptr[w.pos++] = ',';
     }
     first = false;
-    atom(b, item);
+    atom(w, item);
+  }
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = ']';
+}
+
+// =============================================================
+// Size bound: an upper bound on the number of bytes that atom(w, t) writes.
+// Computing it first lets append() reserve the capacity once and then run
+// the whole write chain through an unchecked_writer, without a capacity
+// check before every write. It mirrors the atom() overloads above.
+//
+// Prior related work: jsonifier sizes the document first, counting 6 bytes
+// per string byte, resizes once to that bound plus slack, and writes with
+// no capacity check on each store. to_json() does that resize through
+// std::string::resize_and_overwrite. See serializer.hpp and
+// serialize_impl.hpp in https://github.com/nihilai-collective/Jsonifier
+// and https://nihilai-collective.net/serialization.
+// =============================================================
+namespace bound_detail {
+
+// Whether size_bound covers everything that atom() writes for T: not when a
+// member is serialized by a with<Adapter> serializer, which writes an unknown
+// amount through the string_builder.
+template <class T>
+consteval bool is_bounded() {
+  if constexpr (require_custom_serialization<T>) {
+    return false;
+  } else if constexpr (std::is_same_v<T, std::string> || std::is_same_v<T, std::string_view> ||
+                       std::is_same_v<T, const char *> || std::is_arithmetic_v<T> || std::is_enum_v<T>) {
+    return true;
+  } else if constexpr (concepts::optional_type<T> || concepts::smart_pointer<T>) {
+    return is_bounded<std::remove_cvref_t<decltype(*std::declval<const T &>())>>();
+  } else if constexpr (concepts::string_view_keyed_map<T>) {
+    return is_bounded<std::remove_cvref_t<typename T::mapped_type>>();
+  } else if constexpr (concepts::container_but_not_string<T> || concepts::appendable_containers<T>) {
+    return is_bounded<std::remove_cvref_t<std::ranges::range_value_t<T>>>();
+  } else {
+    bool bounded = true;
+    template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+      if constexpr (annotation_detail::is_serialized_member(dm)) {
+        bounded = bounded && simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::with_t) == std::meta::info{} &&
+                  is_bounded<std::remove_cvref_t<decltype(std::declval<const T &>().[:dm:])>>();
+      }
+    };
+    return bounded;
+  }
+}
+
+template <class T>
+consteval size_t enum_bound() {
+  size_t bound = 20; // the integer fallback
+  template for (constexpr auto enum_val : std::define_static_array(std::meta::enumerators_of(^^T))) {
+    constexpr size_t len = std::char_traits<char>::length(std::define_static_string(
+        constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<enum_val>())));
+    bound = (std::max)(bound, len);
+  };
+  return bound;
+}
+
+template <class T>
+simdjson_really_inline size_t size_bound(const T &t) noexcept;
+
+// Bound for the "key":value pairs of a structure, commas included.
+template <class T>
+simdjson_really_inline size_t fields_bound(const T &t) noexcept {
+  size_t bound = 0;
+  template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+    if constexpr (annotation_detail::is_serialized_member(dm)) {
+      if constexpr (simdjson::detail::has_annotation(dm, ^^simdjson::detail::flatten_tag)) {
+        bound += fields_bound(t.[:dm:]);
+      } else {
+        constexpr size_t rest_key_len = constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<dm>()).size() + 2;
+        bound += rest_key_len + size_bound(t.[:dm:]);
+      }
+    }
+  };
+  return bound;
+}
+
+template <class T>
+simdjson_really_inline size_t size_bound([[maybe_unused]] const T &t) noexcept {
+  if constexpr (std::is_same_v<T, char>) {
+    return 2 + 6;
+  } else if constexpr (std::is_same_v<T, std::string> || std::is_same_v<T, std::string_view> ||
+                       std::is_same_v<T, const char *>) {
+    // Every byte may become \uXXXX, plus the quotes.
+    return 2 + 6 * std::string_view(t).size();
+  } else if constexpr (std::is_same_v<T, bool>) {
+    return 5;
+  } else if constexpr (std::is_floating_point_v<T>) {
+    return simdjson::internal::to_chars_buffer_size;
+  } else if constexpr (std::is_arithmetic_v<T>) {
+    return 20;
+  } else if constexpr (std::is_enum_v<T>) {
+    return enum_bound<T>();
+  } else if constexpr (concepts::optional_type<T> || concepts::smart_pointer<T>) {
+    return t ? size_bound(*t) : 4;
+  } else if constexpr (concepts::string_view_keyed_map<T>) {
+    size_t bound = 2;
+    for (const auto &[key, value] : t) {
+      // comma, quotes, colon
+      bound += 4 + 6 * std::string_view(key).size() + size_bound(value);
+    }
+    return bound;
+  } else if constexpr (concepts::container_but_not_string<T> || concepts::appendable_containers<T>) {
+    using value_type = std::remove_cvref_t<std::ranges::range_value_t<T>>;
+    if constexpr (std::is_arithmetic_v<value_type> && !std::is_same_v<value_type, char>) {
+      // A fixed bound per element: no need to visit them.
+      return 2 + size_t(std::ranges::distance(t)) * (1 + size_bound(value_type{}));
+    } else {
+      size_t bound = 2;
+#if defined(__GNUC__) && !defined(__clang__)
+#pragma GCC novector // a vector loop is slower on short containers
+#endif
+#pragma GCC unroll 4
+      for (const auto &item : t) {
+        bound += 1 + size_bound(item);
+      }
+      return bound;
+    }
+  } else if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag)) {
+    constexpr auto dm = simdjson::detail::transparent_member(^^T);
+    return size_bound(t.[:dm:]);
+  } else {
+    return 2 + fields_bound(t);
   }
-  b.append(']');
 }

-// append functions that delegate to atom functions for primitive types
+} // namespace bound_detail
+
+// Write t through an unchecked writer when its size bound is available,
+// reserving that many bytes first, and through the checked writer otherwise.
+template <class T>
+simdjson_really_inline void append_bounded(string_builder &b, const T &t) {
+  // On 32-bit systems, the bound could overflow: keep the checked writer.
+  if constexpr (sizeof(size_t) >= 8 && bound_detail::is_bounded<T>()) {
+    const size_t bound = bound_detail::size_bound(t) + unchecked_slack;
+    const size_t pos = b.unsafe_position();
+    // The bound is a sum of in-memory sizes times a small constant: it cannot
+    // overflow on a 64-bit system. Be pedantic elsewhere.
+    if (sizeof(size_t) >= 8 || bound <= (std::numeric_limits<size_t>::max)() - pos) {
+      const size_t cap = b.unsafe_capacity();
+      // Grow geometrically so that many small appends stay amortized.
+      if (pos + bound <= cap || b.unsafe_grow((std::max)(cap * 2, pos + bound))) {
+        unchecked_writer w(b.unsafe_data(), pos);
+        atom(w, t);
+        b.unsafe_set_position(w.pos);
+      }
+      return;
+    }
+  }
+  writer w(b);
+  atom(w, t);
+  w.sync();
+}
+
+// append() -- top-level entry. Each overload constructs a stack-local
+// writer, runs atom(w, t) through the inlined call chain, then syncs
+// the local position back into the string_builder.
 template <class T>
   requires(std::is_arithmetic_v<T> && !std::is_same_v<T, char>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  writer w(b);
+  atom(w, t);
+  w.sync();
 }

 template <class T>
@@ -46811,20 +56860,22 @@ template <class T>
            std::is_same_v<T, std::string_view> ||
            std::is_same_v<T, const char *> ||
            std::is_same_v<T, char>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  writer w(b);
+  atom(w, t);
+  w.sync();
 }

 template <concepts::optional_type T>
   requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  append_bounded(b, t);
 }

 template <concepts::smart_pointer T>
   requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  append_bounded(b, t);
 }

 template <concepts::appendable_containers T>
@@ -46832,14 +56883,14 @@ template <concepts::appendable_containers T>
            !concepts::optional_type<T> && !concepts::smart_pointer<T> &&
            !std::is_same_v<T, std::string> &&
            !std::is_same_v<T, std::string_view> && !std::is_same_v<T, const char*> && !require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  append_bounded(b, t);
 }

 template <concepts::string_view_keyed_map T>
   requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  append_bounded(b, t);
 }

 // works for struct
@@ -46853,39 +56904,15 @@ template <class Z>
            !std::is_same_v<Z, std::string_view> &&
            !std::is_same_v<Z, const char*> &&
            !std::is_same_v<Z, char> && !require_custom_serialization<Z>)
-void append(string_builder &b, const Z &z) {
-  int i = 0;
-  b.append('{');
-  template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^Z, std::meta::access_context::unchecked()))) {
-    if (i != 0)
-      b.append(',');
-    constexpr auto key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(dm)));
-    b.append_raw(key);
-    b.append(':');
-    atom(b, z.[:dm:]);
-    i++;
-  };
-  b.append('}');
+simdjson_inline void append(string_builder &b, const Z &z) {
+  append_bounded(b, z);
 }

 // works for container that have begin() and end() iterators
 template <class Z>
   requires(concepts::container_but_not_string<Z> && !require_custom_serialization<Z>)
-void append(string_builder &b, const Z &z) {
-  auto it = z.begin();
-  auto end = z.end();
-  if (it == end) {
-    b.append_raw("[]");
-    return;
-  }
-  b.append('[');
-  atom(b, *it);
-  ++it;
-  for (; it != end; ++it) {
-    b.append(',');
-    atom(b, *it);
-  }
-  b.append(']');
+simdjson_inline void append(string_builder &b, const Z &z) {
+  append_bounded(b, z);
 }

 template <class Z>
@@ -46896,22 +56923,40 @@ void append(string_builder &b, const Z &z) {


 template <class Z>
-simdjson_warn_unused simdjson_result<std::string> to_json_string(const Z &z, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
-  string_builder b(initial_capacity);
-  append(b, z);
-  std::string_view s;
-  if(auto e = b.view().get(s); e) { return e; }
-  return std::string(s);
+simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
+  if constexpr (sizeof(size_t) >= 8 && bound_detail::is_bounded<Z>()) {
+    // Write straight into s, sized by the bound: no intermediate buffer, no copy.
+    // Prior related work: jsonifier's serializeJson resizes once through
+    // resize_and_overwrite (serializer.hpp).
+    (void)initial_capacity;
+    const size_t bound = bound_detail::size_bound(z) + unchecked_slack;
+    auto write = [&z](char *p) noexcept {
+      unchecked_writer w(p, 0);
+      atom(w, z);
+      return w.pos;
+    };
+#if defined(__cpp_lib_string_resize_and_overwrite) && __cpp_lib_string_resize_and_overwrite >= 202110L
+    s.resize_and_overwrite(bound, [&write](char *p, size_t) noexcept { return write(p); });
+#else
+    s.resize(bound);
+    s.resize(write(s.data()));
+#endif
+    return SUCCESS;
+  } else {
+    string_builder b(initial_capacity);
+    append(b, z);
+    std::string_view view;
+    if(auto e = b.view().get(view); e) { return e; }
+    s.assign(view);
+    return SUCCESS;
+  }
 }

 template <class Z>
-simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
-  string_builder b(initial_capacity);
-  append(b, z);
-  std::string_view view;
-  if(auto e = b.view().get(view); e) { return e; }
-  s.assign(view);
-  return SUCCESS;
+simdjson_warn_unused simdjson_result<std::string> to_json_string(const Z &z, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
+  std::string s;
+  if(auto e = to_json(z, s, initial_capacity); e) { return e; }
+  return s;
 }

 template <class Z>
@@ -46924,40 +56969,41 @@ string_builder& operator<<(string_builder& b, const Z& z) {
 template<constevalutil::fixed_string... FieldNames, typename T>
   requires(std::is_class_v<T> && (sizeof...(FieldNames) > 0))
 void extract_from(string_builder &b, const T &obj) {
-  // Helper to check if a field name matches any of the requested fields
-  auto should_extract = [](std::string_view field_name) constexpr -> bool {
-    return ((FieldNames.view() == field_name) || ...);
-  };
-
-  b.append('{');
+  writer w(b);
+  if (!w.ensure(1)) { w.sync(); return; }
+  w.ptr[w.pos++] = '{';
   bool first = true;
-
   // Iterate through all members of T using reflection
-  template for (constexpr auto mem : std::define_static_array(
-      std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
-
+  static constexpr auto members = std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()));
+  template for (constexpr auto mem : members) {
     if constexpr (std::meta::is_public(mem)) {
-      constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
+      static constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));

       // Only serialize this field if it's in our list of requested fields
-      if constexpr (should_extract(key)) {
-        if (!first) {
-          b.append(',');
+      if constexpr (((FieldNames.view() == key) || ...)) {
+        static constexpr auto first_key = std::define_static_string(
+            constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)) + ":");
+        static constexpr auto rest_key = std::define_static_string(
+            std::string(",") + constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)) + ":");
+        constexpr size_t first_key_len = std::char_traits<char>::length(first_key);
+        constexpr size_t rest_key_len = std::char_traits<char>::length(rest_key);
+        if (!w.ensure(rest_key_len)) { w.sync(); return; }
+        if (first) {
+          std::memcpy(w.ptr + w.pos, first_key, first_key_len);
+          w.pos += first_key_len;
+        } else {
+          std::memcpy(w.ptr + w.pos, rest_key, rest_key_len);
+          w.pos += rest_key_len;
         }
         first = false;
-
-        // Serialize the key
-        constexpr auto quoted_key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)));
-        b.append_raw(quoted_key);
-        b.append(':');
-
-        // Serialize the value
-        atom(b, obj.[:mem:]);
+        atom(w, obj.[:mem:]);
       }
     }
   };

-  b.append('}');
+  if (!w.ensure(1)) { w.sync(); return; }
+  w.ptr[w.pos++] = '}';
+  w.sync();
 }

 template<constevalutil::fixed_string... FieldNames, typename T>
@@ -46970,25 +57016,18 @@ simdjson_warn_unused simdjson_result<std::string> extract_from(const T &obj, siz
   return std::string(s);
 }

+SIMDJSON_POP_DISABLE_WARNINGS
+
 } // namespace builder
 } // namespace icelake
 // Alias the function template to 'to' in the global namespace
 template <class Z>
 simdjson_warn_unused simdjson_result<std::string> to_json(const Z &z, size_t initial_capacity = icelake::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
-  icelake::builder::string_builder b(initial_capacity);
-  icelake::builder::append(b, z);
-  std::string_view s;
-  if(auto e = b.view().get(s); e) { return e; }
-  return std::string(s);
+  return icelake::builder::to_json_string(z, initial_capacity);
 }
 template <class Z>
 simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = icelake::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
-  icelake::builder::string_builder b(initial_capacity);
-  icelake::builder::append(b, z);
-  std::string_view view;
-  if(auto e = b.view().get(view); e) { return e; }
-  s.assign(view);
-  return SUCCESS;
+  return icelake::builder::to_json(z, s, initial_capacity);
 }
 // Global namespace function for extract_from
 template<constevalutil::fixed_string... FieldNames, typename T>
@@ -47134,6 +57173,7 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
 /* including simdjson/generic/builder/json_string_builder-inl.h for icelake: #include "simdjson/generic/builder/json_string_builder-inl.h" */
 /* begin file simdjson/generic/builder/json_string_builder-inl.h for icelake */
 #include <array>
+#include <cmath>
 #include <cstring>
 #include <limits>
 #include <type_traits>
@@ -47168,6 +57208,11 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
 #define SIMDJSON_EXPERIMENTAL_HAS_LSX 1
 #endif
 #endif
+#if defined(__loongarch_asx)
+#ifndef SIMDJSON_EXPERIMENTAL_HAS_LASX
+#define SIMDJSON_EXPERIMENTAL_HAS_LASX 1
+#endif
+#endif
 #if defined(__riscv_v_intrinsic) && __riscv_v_intrinsic >= 11000 &&            \
     defined(__riscv_vector)
 #ifndef SIMDJSON_EXPERIMENTAL_HAS_RVV
@@ -47187,6 +57232,9 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
 #endif
 #if SIMDJSON_EXPERIMENTAL_HAS_SSE2
 #include <emmintrin.h>
+#if defined(__AVX2__)
+#include <immintrin.h>
+#endif
 #ifdef _MSC_VER
 #include <intrin.h>
 #endif
@@ -47194,6 +57242,9 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
 #if SIMDJSON_EXPERIMENTAL_HAS_LSX
 #include <lsxintrin.h>
 #endif
+#if SIMDJSON_EXPERIMENTAL_HAS_LASX
+#include <lasxintrin.h>
+#endif
 #if SIMDJSON_EXPERIMENTAL_HAS_RVV
 #include <riscv_vector.h>
 #endif
@@ -47243,105 +57294,6 @@ inline bool has_json_escapable_byte(uint64_t x) {

 **/

-SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline bool
-simple_needs_escaping(std::string_view v) {
-  for (char c : v) {
-    // a table lookup is faster than a series of comparisons
-    if (json_quotable_character[static_cast<uint8_t>(c)]) {
-      return true;
-    }
-  }
-  return false;
-}
-
-#if SIMDJSON_EXPERIMENTAL_HAS_NEON
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
-  if (view.size() < 16) {
-    return simple_needs_escaping(view);
-  }
-  size_t i = 0;
-  uint8x16_t running = vdupq_n_u8(0);
-  uint8x16_t v34 = vdupq_n_u8(34);
-  uint8x16_t v92 = vdupq_n_u8(92);
-
-  for (; i + 15 < view.size(); i += 16) {
-    uint8x16_t word = vld1q_u8((const uint8_t *)view.data() + i);
-    running = vorrq_u8(running, vceqq_u8(word, v34));
-    running = vorrq_u8(running, vceqq_u8(word, v92));
-    running = vorrq_u8(running, vcltq_u8(word, vdupq_n_u8(32)));
-  }
-  if (i < view.size()) {
-    uint8x16_t word =
-        vld1q_u8((const uint8_t *)view.data() + view.length() - 16);
-    running = vorrq_u8(running, vceqq_u8(word, v34));
-    running = vorrq_u8(running, vceqq_u8(word, v92));
-    running = vorrq_u8(running, vcltq_u8(word, vdupq_n_u8(32)));
-  }
-  return vmaxvq_u32(vreinterpretq_u32_u8(running)) != 0;
-}
-#elif SIMDJSON_EXPERIMENTAL_HAS_SSE2
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
-  if (view.size() < 16) {
-    return simple_needs_escaping(view);
-  }
-  size_t i = 0;
-  __m128i running = _mm_setzero_si128();
-  for (; i + 15 < view.size(); i += 16) {
-
-    __m128i word =
-        _mm_loadu_si128(reinterpret_cast<const __m128i *>(view.data() + i));
-    running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(34)));
-    running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(92)));
-    running = _mm_or_si128(
-        running, _mm_cmpeq_epi8(_mm_subs_epu8(word, _mm_set1_epi8(31)),
-                                _mm_setzero_si128()));
-  }
-  if (i < view.size()) {
-    __m128i word = _mm_loadu_si128(
-        reinterpret_cast<const __m128i *>(view.data() + view.length() - 16));
-    running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(34)));
-    running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(92)));
-    running = _mm_or_si128(
-        running, _mm_cmpeq_epi8(_mm_subs_epu8(word, _mm_set1_epi8(31)),
-                                _mm_setzero_si128()));
-  }
-  return _mm_movemask_epi8(running) != 0;
-}
-#elif SIMDJSON_EXPERIMENTAL_HAS_PPC64
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
-  if (view.size() < 16) {
-    return simple_needs_escaping(view);
-  }
-  size_t i = 0;
-  __vector unsigned char running = vec_splats((unsigned char)0);
-  __vector unsigned char v34 = vec_splats((unsigned char)34);
-  __vector unsigned char v92 = vec_splats((unsigned char)92);
-  __vector unsigned char v32 = vec_splats((unsigned char)32);
-
-  for (; i + 15 < view.size(); i += 16) {
-    __vector unsigned char word =
-        vec_vsx_ld(0, reinterpret_cast<const unsigned char *>(view.data() + i));
-    running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v34));
-    running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v92));
-    running = vec_or(running,
-        (__vector unsigned char)vec_cmplt(word, v32));
-  }
-  if (i < view.size()) {
-    __vector unsigned char word = vec_vsx_ld(
-        0, reinterpret_cast<const unsigned char *>(view.data() + view.length() - 16));
-    running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v34));
-    running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v92));
-    running = vec_or(running,
-        (__vector unsigned char)vec_cmplt(word, v32));
-  }
-  return !vec_all_eq(running, vec_splats((unsigned char)0));
-}
-#else
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
-  return simple_needs_escaping(view);
-}
-#endif
-
 // Scalar fallback for finding next quotable character
 SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline size_t
 find_next_json_quotable_character_scalar(const std::string_view view,
@@ -47432,6 +57384,51 @@ find_next_json_quotable_character(const std::string_view view,
   size_t current = len - remaining;
   return find_next_json_quotable_character_scalar(view, current);
 }
+#elif SIMDJSON_EXPERIMENTAL_HAS_LASX
+simdjson_inline size_t
+find_next_json_quotable_character(const std::string_view view,
+                                  size_t location) noexcept {
+  const size_t len = view.size();
+  const uint8_t *ptr =
+      reinterpret_cast<const uint8_t *>(view.data()) + location;
+  size_t remaining = len - location;
+
+  // SIMD constants for characters requiring escape
+  __m256i v34 = __lasx_xvreplgr2vr_b(34);  // '"'
+  __m256i v92 = __lasx_xvreplgr2vr_b(92);  // '\\'
+  __m256i v32 = __lasx_xvreplgr2vr_b(32);  // control char threshold
+
+  while (remaining >= 32) {
+    __m256i word = __lasx_xvld(ptr, 0);
+
+    // Check for quotable characters: '"', '\\', or control chars (< 32)
+    __m256i needs_escape = __lasx_xvseq_b(word, v34);
+    needs_escape = __lasx_xvor_v(needs_escape, __lasx_xvseq_b(word, v92));
+    needs_escape = __lasx_xvor_v(needs_escape, __lasx_xvslt_bu(word, v32));
+
+    if (!__lasx_xbz_v(needs_escape)) {
+      // Found a quotable character - locate it via the four 64-bit lanes
+      uint64_t lane0 = __lasx_xvpickve2gr_du(needs_escape, 0);
+      uint64_t lane1 = __lasx_xvpickve2gr_du(needs_escape, 1);
+      uint64_t lane2 = __lasx_xvpickve2gr_du(needs_escape, 2);
+      uint64_t lane3 = __lasx_xvpickve2gr_du(needs_escape, 3);
+      size_t offset = ptr - reinterpret_cast<const uint8_t *>(view.data());
+      if (lane0 != 0) {
+        return offset + trailing_zeroes(lane0) / 8;
+      } else if (lane1 != 0) {
+        return offset + 8 + trailing_zeroes(lane1) / 8;
+      } else if (lane2 != 0) {
+        return offset + 16 + trailing_zeroes(lane2) / 8;
+      } else {
+        return offset + 24 + trailing_zeroes(lane3) / 8;
+      }
+    }
+    ptr += 32;
+    remaining -= 32;
+  }
+  size_t current = len - remaining;
+  return find_next_json_quotable_character_scalar(view, current);
+}
 #elif SIMDJSON_EXPERIMENTAL_HAS_LSX
 simdjson_inline size_t
 find_next_json_quotable_character(const std::string_view view,
@@ -47587,6 +57584,254 @@ SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline void escape_json_char(char c, char *&o
   }
 }

+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2 || SIMDJSON_EXPERIMENTAL_HAS_NEON
+#define SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE 1
+#endif
+
+#if SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2
+
+using escape_vector = __m128i;
+// Mask bits that each input byte contributes to escape_bitmask().
+static constexpr unsigned escape_mask_bits = 1;
+
+simdjson_inline escape_vector escape_load16(const uint8_t *p) noexcept {
+  return _mm_loadu_si128(reinterpret_cast<const __m128i *>(p));
+}
+
+simdjson_inline void escape_store16(char *out, escape_vector v) noexcept {
+  _mm_storeu_si128(reinterpret_cast<__m128i *>(out), v);
+}
+
+// Builds a vector whose bytes 0..7 come from a and bytes 8..15 from b.
+simdjson_inline escape_vector escape_load8x2(const uint8_t *a,
+                                             const uint8_t *b) noexcept {
+  return _mm_unpacklo_epi64(
+      _mm_loadl_epi64(reinterpret_cast<const __m128i *>(a)),
+      _mm_loadl_epi64(reinterpret_cast<const __m128i *>(b)));
+}
+
+// Builds a vector whose bytes 0..3 come from a and bytes 4..7 from b. The
+// upper half is unspecified; callers only look at the low eight bits.
+simdjson_inline escape_vector escape_load4x2(const uint8_t *a,
+                                             const uint8_t *b) noexcept {
+  int32_t a32, b32;
+  memcpy(&a32, a, 4);
+  memcpy(&b32, b, 4);
+  return _mm_unpacklo_epi32(_mm_cvtsi32_si128(a32), _mm_cvtsi32_si128(b32));
+}
+
+// Sets every bit of byte k when byte k of v is a quotable character ('"',
+// '\\' or a control character).
+simdjson_inline escape_vector escape_flags(escape_vector v) noexcept {
+  const __m128i v34 = _mm_set1_epi8(34); // '"'
+  const __m128i v92 = _mm_set1_epi8(92); // '\\'
+  const __m128i v31 = _mm_set1_epi8(31); // for control char detection
+  __m128i needs_escape = _mm_cmpeq_epi8(v, v34);
+  needs_escape = _mm_or_si128(needs_escape, _mm_cmpeq_epi8(v, v92));
+  return _mm_or_si128(
+      needs_escape, _mm_cmpeq_epi8(_mm_subs_epu8(v, v31), _mm_setzero_si128()));
+}
+
+// True when any byte needs escaping. Kept separate from escape_bitmask
+// because some instruction sets can answer it without leaving the vector
+// register file; on SSE2 the compiler folds the two together.
+simdjson_inline bool escape_any(escape_vector flags) noexcept {
+  return _mm_movemask_epi8(flags) != 0;
+}
+
+// A 16-bit mask with bit k set when byte k needed escaping.
+simdjson_inline uint64_t escape_bitmask(escape_vector flags) noexcept {
+  return uint64_t(uint32_t(_mm_movemask_epi8(flags)));
+}
+
+#else // SIMDJSON_EXPERIMENTAL_HAS_NEON
+
+using escape_vector = uint8x16_t;
+static constexpr unsigned escape_mask_bits = 4;
+
+simdjson_inline escape_vector escape_load16(const uint8_t *p) noexcept {
+  return vld1q_u8(p);
+}
+
+simdjson_inline void escape_store16(char *out, escape_vector v) noexcept {
+  vst1q_u8(reinterpret_cast<uint8_t *>(out), v);
+}
+
+simdjson_inline escape_vector escape_load8x2(const uint8_t *a,
+                                             const uint8_t *b) noexcept {
+  uint64_t a64, b64;
+  memcpy(&a64, a, 8);
+  memcpy(&b64, b, 8);
+  return vreinterpretq_u8_u64(vsetq_lane_u64(b64, vdupq_n_u64(a64), 1));
+}
+
+simdjson_inline escape_vector escape_load4x2(const uint8_t *a,
+                                             const uint8_t *b) noexcept {
+  uint32_t a32, b32;
+  memcpy(&a32, a, 4);
+  memcpy(&b32, b, 4);
+  return vreinterpretq_u8_u32(vsetq_lane_u32(b32, vdupq_n_u32(a32), 1));
+}
+
+simdjson_inline escape_vector escape_flags(escape_vector v) noexcept {
+  uint8x16_t needs_escape = vceqq_u8(v, vdupq_n_u8(34));              // '"'
+  needs_escape = vorrq_u8(needs_escape, vceqq_u8(v, vdupq_n_u8(92))); // '\\'
+  return vorrq_u8(needs_escape, vcltq_u8(v, vdupq_n_u8(32)));
+}
+
+simdjson_inline bool escape_any(escape_vector flags) noexcept {
+  return vmaxvq_u32(vreinterpretq_u32_u8(flags)) != 0;
+}
+
+// Four bits per byte rather than one: escape_block and the tail paths scale
+// their shifts by escape_mask_bits to match.
+simdjson_inline uint64_t escape_bitmask(escape_vector flags) noexcept {
+  uint8x8_t narrowed = vshrn_n_u16(vreinterpretq_u16_u8(flags), 4);
+  return vget_lane_u64(vreinterpret_u64_u8(narrowed), 0);
+}
+
+#endif // instruction set selection
+
+// The escape bitmask of a 16-byte block.
+simdjson_inline uint64_t escape_mask(escape_vector v) noexcept {
+  return escape_bitmask(escape_flags(v));
+}
+
+// Copies n bytes with n < 16, using overlapping loads and stores. It never
+// reads more than n bytes from src, nor writes more than n bytes to dst.
+simdjson_inline void copy_lt16(char *dst, const uint8_t *src,
+                               size_t n) noexcept {
+  if (n >= 8) {
+    memcpy(dst, src, 8);
+    memcpy(dst + n - 8, src + n - 8, 8);
+  } else if (n >= 4) {
+    memcpy(dst, src, 4);
+    memcpy(dst + n - 4, src + n - 4, 4);
+  } else if (n > 0) {
+    dst[0] = char(src[0]);
+    dst[n >> 1] = char(src[n >> 1]);
+    dst[n - 1] = char(src[n - 1]);
+  }
+}
+
+// Escapes the bytes of src in the range [i, blockend), given that m is the
+// (non-zero) escape mask for that range: the escape_mask_bits-wide lane of m
+// at byte k is non-zero when src[i + k] requires escaping. Returns the updated
+// output pointer.
+simdjson_never_inline char *escape_block(const uint8_t *src, char *out,
+                                         size_t i, size_t blockend,
+                                         uint64_t m) noexcept {
+  constexpr uint64_t lane = (uint64_t(1) << escape_mask_bits) - 1;
+
+  size_t pos = i; // first byte not yet copied
+  while (m) {
+    const size_t tz = trailing_zeroes(m);
+    const size_t next = i + tz / escape_mask_bits;
+    // Copy the run of safe bytes that precedes this escape.
+    copy_lt16(out, src + pos, next - pos);
+    out += next - pos;
+    escape_json_char(char(src[next]), out);
+    pos = next + 1;
+    m &= ~(lane << tz);
+  }
+  // Copy whatever follows the last escape.
+  copy_lt16(out, src + pos, blockend - pos);
+  return out + (blockend - pos);
+}
+
+// Writes the escaped version of input to out, returning the number of bytes
+// written.
+simdjson_really_inline size_t write_string_escaped(const std::string_view input, char *out) {
+  const size_t len = input.size();
+  const uint8_t *src = reinterpret_cast<const uint8_t *>(input.data());
+  const char *const initout = out;
+
+  size_t i = 0;
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2 && defined(__AVX2__)
+  while (i + 32 <= len) {
+    const __m256i word = _mm256_loadu_si256(reinterpret_cast<const __m256i *>(src + i));
+    const __m256i flags = _mm256_or_si256(
+        _mm256_or_si256(_mm256_cmpeq_epi8(word, _mm256_set1_epi8(34)),   // '"'
+                        _mm256_cmpeq_epi8(word, _mm256_set1_epi8(92))),  // '\\'
+        _mm256_cmpeq_epi8(_mm256_subs_epu8(word, _mm256_set1_epi8(31)),
+                          _mm256_setzero_si256()));                      // control
+    const uint32_t mask = uint32_t(_mm256_movemask_epi8(flags));
+    if (simdjson_likely(mask == 0)) {
+      _mm256_storeu_si256(reinterpret_cast<__m256i *>(out), word);
+      out += 32;
+    } else {
+      for (size_t half = 0; half < 32; half += 16) {
+        const uint64_t m = (mask >> half) & 0xFFFF;
+        if (m == 0) {
+          escape_store16(out, escape_load16(src + i + half));
+          out += 16;
+        } else {
+          out = escape_block(src, out, i + half, i + half + 16, m);
+        }
+      }
+    }
+    i += 32;
+  }
+#endif
+  while (i + 16 <= len) {
+    escape_vector word = escape_load16(src + i);
+    escape_vector flags = escape_flags(word);
+    if (simdjson_likely(!escape_any(flags))) {
+      escape_store16(out, word);
+      out += 16;
+    } else {
+      out = escape_block(src, out, i, i + 16, escape_bitmask(flags));
+    }
+    i += 16;
+  }
+  if (i < len) {
+    const size_t rem = len - i;
+    uint64_t m;
+    if (len >= 16) {
+      // The last 16 bytes of the input are in bounds. Bit k of that block's
+      // mask belongs to input position len - 16 + k, so shift it down to align
+      // bit 0 with position i.
+      m = escape_mask(escape_load16(src + len - 16)) >>
+          (escape_mask_bits * (16 - rem));
+    } else if (len >= 8) {
+      // Here i == 0 and rem == len. Two overlapping 8-byte loads cover
+      // [0, 8) and [len - 8, len), which is the whole input since len < 16.
+      uint64_t mm = escape_mask(escape_load8x2(src, src + len - 8));
+      constexpr uint64_t low8 = (uint64_t(1) << (escape_mask_bits * 8)) - 1;
+      m = (mm & low8) |
+          ((mm >> (escape_mask_bits * 8)) << (escape_mask_bits * (len - 8)));
+    } else if (len >= 4) {
+      // Same idea with two overlapping 4-byte loads.
+      uint64_t mm = escape_mask(escape_load4x2(src, src + len - 4));
+      constexpr uint64_t low4 = (uint64_t(1) << (escape_mask_bits * 4)) - 1;
+      m = (mm & low4) | (((mm >> (escape_mask_bits * 4)) & low4)
+                         << (escape_mask_bits * (len - 4)));
+    } else {
+      // Fewer than 4 bytes: at most three table lookups, no need for SIMD.
+      for (size_t k = 0; k < len; k++) {
+        uint8_t c = src[k];
+        if (json_quotable_character[c]) {
+          escape_json_char(char(c), out);
+        } else {
+          *out++ = char(c);
+        }
+      }
+      return size_t(out - initout);
+    }
+    if (m == 0) {
+      copy_lt16(out, src + i, rem);
+      out += rem;
+    } else {
+      out = escape_block(src, out, i, len, m);
+    }
+  }
+  return size_t(out - initout);
+}
+
+#else // SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
 // Writes the escaped version of input to out, returning the number of bytes
 // written. Uses SIMD position finding to locate quotable characters efficiently.
 inline size_t write_string_escaped(const std::string_view input, char *out) {
@@ -47616,9 +57861,13 @@ inline size_t write_string_escaped(const std::string_view input, char *out) {
     escape_json_char(input[location], out);
     location += 1;
   }
-  return out - initout;
+  return size_t(out - initout);
 }

+#endif // SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+#undef SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
+
 simdjson_inline string_builder::string_builder(size_t initial_capacity)
     : buffer(new(std::nothrow) char[initial_capacity]), position(0),
       capacity(buffer.get() != nullptr ? initial_capacity : 0),
@@ -47641,7 +57890,7 @@ simdjson_inline bool string_builder::capacity_check(size_t upcoming_bytes) {
   return is_valid;
 }

-simdjson_inline void string_builder::grow_buffer(size_t desired_capacity) {
+inline void string_builder::grow_buffer(size_t desired_capacity) {
   if (!is_valid) {
     return;
   }
@@ -47697,81 +57946,136 @@ simdjson_inline void string_builder::clear() noexcept {

 namespace internal {

-template <typename number_type, typename = typename std::enable_if<
-                                    std::is_unsigned<number_type>::value>::type>
-simdjson_really_inline int int_log2(number_type x) {
-  return 63 - leading_zeroes(uint64_t(x) | 1);
-}
-
-simdjson_really_inline int fast_digit_count_32(uint32_t x) {
-  static uint64_t table[] = {
-      4294967296,  8589934582,  8589934582,  8589934582,  12884901788,
-      12884901788, 12884901788, 17179868184, 17179868184, 17179868184,
-      21474826480, 21474826480, 21474826480, 21474826480, 25769703776,
-      25769703776, 25769703776, 30063771072, 30063771072, 30063771072,
-      34349738368, 34349738368, 34349738368, 34349738368, 38554705664,
-      38554705664, 38554705664, 41949672960, 41949672960, 41949672960,
-      42949672960, 42949672960};
-  return uint32_t((x + table[int_log2(x)]) >> 32);
-}
-
-simdjson_really_inline int fast_digit_count_64(uint64_t x) {
-  static uint64_t table[] = {9,
-                             99,
-                             999,
-                             9999,
-                             99999,
-                             999999,
-                             9999999,
-                             99999999,
-                             999999999,
-                             9999999999,
-                             99999999999,
-                             999999999999,
-                             9999999999999,
-                             99999999999999,
-                             999999999999999ULL,
-                             9999999999999999ULL,
-                             99999999999999999ULL,
-                             999999999999999999ULL,
-                             9999999999999999999ULL};
-  int y = (19 * int_log2(x) >> 6);
-  y += x > table[y];
-  return y + 1;
-}
-
-template <typename number_type, typename = typename std::enable_if<
-                                    std::is_unsigned<number_type>::value>::type>
-simdjson_really_inline size_t digit_count(number_type v) noexcept {
-  static_assert(sizeof(number_type) == 8 || sizeof(number_type) == 4 ||
-                    sizeof(number_type) == 2 || sizeof(number_type) == 1,
-                "We only support 8-bit, 16-bit, 32-bit and 64-bit numbers");
-  SIMDJSON_IF_CONSTEXPR(sizeof(number_type) <= 4) {
-    return fast_digit_count_32(static_cast<uint32_t>(v));
+// Integer to decimal: James Edward Anhalt III's algorithm
+static const char jeaiii_dd[201] =
+    "00010203040506070809101112131415161718192021222324252627282930313233343536373839"
+    "40414243444546474849505152535455565758596061626364656667686970717273747576777879"
+    "8081828384858687888990919293949596979899";
+static const char jeaiii_fd[201] =
+    "0\0" "1\0" "2\0" "3\0" "4\0" "5\0" "6\0" "7\0" "8\0" "9\0"
+    "10111213141516171819202122232425262728293031323334353637383940414243444546474849"
+    "50515253545556575859606162636465666768697071727374757677787980818283848586878889"
+    "90919293949596979899";
+
+simdjson_really_inline void jeaiii_write_dd(char *p, uint64_t k) noexcept {
+  std::memcpy(p, &jeaiii_dd[2 * k], 2);
+}
+simdjson_really_inline void jeaiii_write_fd(char *p, uint64_t k) noexcept {
+  std::memcpy(p, &jeaiii_fd[2 * k], 2);
+}
+
+// Caller guarantees n < 10^8. Writes 1 to 8 digits.
+simdjson_really_inline char *jeaiii_lt1e8(char *b, uint32_t n) noexcept {
+  constexpr uint64_t mask24 = (uint64_t(1) << 24) - 1;
+  constexpr uint64_t mask32 = (uint64_t(1) << 32) - 1;
+  if (n < 100) {
+    jeaiii_write_fd(b, n);
+    return n < 10 ? b + 1 : b + 2;
+  }
+  if (n < 1000000) {
+    if (n < 10000) {
+      const uint32_t f0 = uint32_t(10 * (1 << 24) / 1e3 + 1) * n;
+      jeaiii_write_fd(b, f0 >> 24);
+      b -= n < 1000;
+      const uint32_t f2 = uint32_t(f0 & mask24) * 100;
+      jeaiii_write_dd(b + 2, f2 >> 24);
+      return b + 4;
+    }
+    const uint64_t f0 = uint64_t(10 * (1ull << 32) / 1e5 + 1) * n;
+    jeaiii_write_fd(b, f0 >> 32);
+    b -= n < 100000;
+    const uint64_t f2 = (f0 & mask32) * 100;
+    jeaiii_write_dd(b + 2, f2 >> 32);
+    const uint64_t f4 = (f2 & mask32) * 100;
+    jeaiii_write_dd(b + 4, f4 >> 32);
+    return b + 6;
+  }
+  const uint64_t f0 = uint64_t(10 * (1ull << 48) / 1e7 + 1) * n >> 16;
+  jeaiii_write_fd(b, f0 >> 32);
+  b -= n < 10000000;
+  const uint64_t f2 = (f0 & mask32) * 100;
+  jeaiii_write_dd(b + 2, f2 >> 32);
+  const uint64_t f4 = (f2 & mask32) * 100;
+  jeaiii_write_dd(b + 4, f4 >> 32);
+  const uint64_t f6 = (f4 & mask32) * 100;
+  jeaiii_write_dd(b + 6, f6 >> 32);
+  return b + 8;
+}
+
+// Caller guarantees z < 10^8. Always writes exactly 8 digits.
+simdjson_really_inline char *jeaiii_8_digits(char *b, uint32_t z) noexcept {
+  constexpr uint64_t mask32 = (uint64_t(1) << 32) - 1;
+  const uint64_t f0 = (uint64_t((1ull << 48) / 1e6 + 1) * z >> 16) + 1;
+  jeaiii_write_dd(b, f0 >> 32);
+  const uint64_t f2 = (f0 & mask32) * 100;
+  jeaiii_write_dd(b + 2, f2 >> 32);
+  const uint64_t f4 = (f2 & mask32) * 100;
+  jeaiii_write_dd(b + 4, f4 >> 32);
+  const uint64_t f6 = (f4 & mask32) * 100;
+  jeaiii_write_dd(b + 6, f6 >> 32);
+  return b + 8;
+}
+
+// Caller guarantees 10^8 <= n < 2^32. Writes 9 or 10 digits.
+simdjson_really_inline char *jeaiii_9_or_10(char *b, uint64_t n) noexcept {
+  constexpr uint64_t mask57 = (uint64_t(1) << 57) - 1;
+  const uint64_t f0 = uint64_t(10 * (1ull << 57) / 1e9 + 1) * n;
+  jeaiii_write_fd(b, f0 >> 57);
+  b -= n < 1000000000;
+  const uint64_t f2 = (f0 & mask57) * 100;
+  jeaiii_write_dd(b + 2, f2 >> 57);
+  const uint64_t f4 = (f2 & mask57) * 100;
+  jeaiii_write_dd(b + 4, f4 >> 57);
+  const uint64_t f6 = (f4 & mask57) * 100;
+  jeaiii_write_dd(b + 6, f6 >> 57);
+  const uint64_t f8 = (f6 & mask57) * 100;
+  jeaiii_write_dd(b + 8, f8 >> 57);
+  return b + 10;
+}
+
+simdjson_really_inline char *write_uint_jeaiii(char *b, uint64_t n) noexcept {
+  if (n < 100000000) {
+    return jeaiii_lt1e8(b, uint32_t(n));
+  }
+  if (n < (uint64_t(1) << 32)) {
+    return jeaiii_9_or_10(b, n);
+  }
+  // At least 10 digits: the low 8 digits, and 2 to 12 digits above them.
+  const uint32_t z = uint32_t(n % 100000000);
+  uint64_t u = n / 100000000;
+  if (u < 100000000) {
+    // u has 2 to 8 digits (if u < 10, n would be below 2^32).
+    b = jeaiii_lt1e8(b, uint32_t(u));
+  } else if (u < (uint64_t(1) << 32)) {
+    b = jeaiii_9_or_10(b, u);
+  } else {
+    // u has 11 or 12 digits: split off 8 more.
+    const uint32_t y = uint32_t(u % 100000000);
+    u /= 100000000;
+    b = jeaiii_lt1e8(b, uint32_t(u)); // 3 or 4 digits
+    b = jeaiii_8_digits(b, y);
   }
-  else {
-    return fast_digit_count_64(static_cast<uint64_t>(v));
-  }
-}
-static const char decimal_table[200] = {
-    0x30, 0x30, 0x30, 0x31, 0x30, 0x32, 0x30, 0x33, 0x30, 0x34, 0x30, 0x35,
-    0x30, 0x36, 0x30, 0x37, 0x30, 0x38, 0x30, 0x39, 0x31, 0x30, 0x31, 0x31,
-    0x31, 0x32, 0x31, 0x33, 0x31, 0x34, 0x31, 0x35, 0x31, 0x36, 0x31, 0x37,
-    0x31, 0x38, 0x31, 0x39, 0x32, 0x30, 0x32, 0x31, 0x32, 0x32, 0x32, 0x33,
-    0x32, 0x34, 0x32, 0x35, 0x32, 0x36, 0x32, 0x37, 0x32, 0x38, 0x32, 0x39,
-    0x33, 0x30, 0x33, 0x31, 0x33, 0x32, 0x33, 0x33, 0x33, 0x34, 0x33, 0x35,
-    0x33, 0x36, 0x33, 0x37, 0x33, 0x38, 0x33, 0x39, 0x34, 0x30, 0x34, 0x31,
-    0x34, 0x32, 0x34, 0x33, 0x34, 0x34, 0x34, 0x35, 0x34, 0x36, 0x34, 0x37,
-    0x34, 0x38, 0x34, 0x39, 0x35, 0x30, 0x35, 0x31, 0x35, 0x32, 0x35, 0x33,
-    0x35, 0x34, 0x35, 0x35, 0x35, 0x36, 0x35, 0x37, 0x35, 0x38, 0x35, 0x39,
-    0x36, 0x30, 0x36, 0x31, 0x36, 0x32, 0x36, 0x33, 0x36, 0x34, 0x36, 0x35,
-    0x36, 0x36, 0x36, 0x37, 0x36, 0x38, 0x36, 0x39, 0x37, 0x30, 0x37, 0x31,
-    0x37, 0x32, 0x37, 0x33, 0x37, 0x34, 0x37, 0x35, 0x37, 0x36, 0x37, 0x37,
-    0x37, 0x38, 0x37, 0x39, 0x38, 0x30, 0x38, 0x31, 0x38, 0x32, 0x38, 0x33,
-    0x38, 0x34, 0x38, 0x35, 0x38, 0x36, 0x38, 0x37, 0x38, 0x38, 0x38, 0x39,
-    0x39, 0x30, 0x39, 0x31, 0x39, 0x32, 0x39, 0x33, 0x39, 0x34, 0x39, 0x35,
-    0x39, 0x36, 0x39, 0x37, 0x39, 0x38, 0x39, 0x39,
-};
+  return jeaiii_8_digits(b, z);
+}
+
+// Writes v at p, which must have to_chars_buffer_size bytes available, and
+// returns the end of what was written.
+simdjson_inline char *write_double(char *p, double v) noexcept {
+#if SIMDJSON_ENABLE_NAN_INF
+  if (simdjson_unlikely(!std::isfinite(v))) {
+    if (std::isnan(v)) {
+      std::memcpy(p, "NaN", 3);
+      return p + 3;
+    }
+    if (v < 0) {
+      *p++ = '-';
+    }
+    std::memcpy(p, "Infinity", 8);
+    return p + 8;
+  }
+#endif
+  return simdjson::internal::to_chars(p, nullptr, v);
+}
 } // namespace internal

 template <typename number_type, typename>
@@ -47799,87 +58103,62 @@ simdjson_inline void string_builder::append(number_type v) noexcept {
     }
   }
   else SIMDJSON_IF_CONSTEXPR(std::is_unsigned<number_type>::value) {
-    // Process 4 digits at a time instead of 2, reducing store operations
-    // and divisions by approximately half for large numbers.
     constexpr size_t max_number_size = 20;
     if (capacity_check(max_number_size)) {
       using unsigned_type = typename std::make_unsigned<number_type>::type;
-      unsigned_type pv = static_cast<unsigned_type>(v);
-      size_t dc = internal::digit_count(pv);
-      char *write_pointer = buffer.get() + position + dc - 1;
-
-      // Process 4 digits per iteration for large numbers
-      while (pv >= 10000) {
-        unsigned_type q = pv / 10000;
-        unsigned_type r = pv % 10000;
-        unsigned_type r_hi = r / 100;  // High 2 digits of remainder
-        unsigned_type r_lo = r % 100;  // Low 2 digits of remainder
-        // Write low 2 digits first (rightmost), then high 2 digits
-        memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
-        memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
-        write_pointer -= 4;
-        pv = q;
-      }
-
-      // Handle remaining 1-4 digits with original 2-digit loop
-      while (pv >= 100) {
-        memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
-        write_pointer -= 2;
-        pv /= 100;
-      }
-      if (pv >= 10) {
-        *write_pointer-- = char('0' + (pv % 10));
-        pv /= 10;
-      }
-      *write_pointer = char('0' + pv);
-      position += dc;
+      char* end = internal::write_uint_jeaiii(
+          buffer.get() + position,
+          static_cast<uint64_t>(static_cast<unsigned_type>(v)));
+      position = end - buffer.get();
     }
   }
   else SIMDJSON_IF_CONSTEXPR(std::is_integral<number_type>::value) {
-    // Same 4-digit batching as unsigned path for signed integers
+    // 19 digits (max abs value of int64_t) + optional minus sign.
     constexpr size_t max_number_size = 20;
     if (capacity_check(max_number_size)) {
       using unsigned_type = typename std::make_unsigned<number_type>::type;
       bool negative = v < 0;
-      unsigned_type pv = static_cast<unsigned_type>(v);
-      if (negative) {
-        pv = 0 - pv; // the 0 is for Microsoft
-      }
-      size_t dc = internal::digit_count(pv);
-      // by always writing the minus sign, we avoid the branch.
+      // 0 - pv (rather than -pv) avoids an MSVC unary-minus warning.
+      unsigned_type pv = negative
+          ? unsigned_type(0) - static_cast<unsigned_type>(v)
+          : static_cast<unsigned_type>(v);
+      // Branchless: always write '-', advance only if negative.
       buffer.get()[position] = '-';
-      position += negative ? 1 : 0;
-      char *write_pointer = buffer.get() + position + dc - 1;
-
-      // Process 4 digits per iteration for large numbers
-      while (pv >= 10000) {
-        unsigned_type q = pv / 10000;
-        unsigned_type r = pv % 10000;
-        unsigned_type r_hi = r / 100;
-        unsigned_type r_lo = r % 100;
-        memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
-        memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
-        write_pointer -= 4;
-        pv = q;
-      }
-
-      // Handle remaining 1-4 digits
-      while (pv >= 100) {
-        memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
-        write_pointer -= 2;
-        pv /= 100;
-      }
-      if (pv >= 10) {
-        *write_pointer-- = char('0' + (pv % 10));
-        pv /= 10;
-      }
-      *write_pointer = char('0' + pv);
-      position += dc;
+      position += negative;
+      char* end = internal::write_uint_jeaiii(
+          buffer.get() + position, static_cast<uint64_t>(pv));
+      position = end - buffer.get();
     }
   }
   else SIMDJSON_IF_CONSTEXPR(std::is_floating_point<number_type>::value) {
-    constexpr size_t max_number_size = 24;
+    // Must reserve to_chars_buffer_size (40): only ~24 chars are emitted,
+    // but to_chars over-writes with fixed-size 16/17-byte copies so the
+    // compiler can inline mem* (see simdjson::internal::to_chars_buffer_size).
+    constexpr size_t max_number_size = simdjson::internal::to_chars_buffer_size;
     if (capacity_check(max_number_size)) {
+#if SIMDJSON_ENABLE_NAN_INF
+      // Check if the input might be NaN or infinity
+      if (simdjson_unlikely(!std::isfinite(v))) {
+        if (std::isnan(v)) {
+          constexpr char nan_literal[] = "NaN";
+          constexpr size_t nan_len = sizeof(nan_literal) - 1;
+
+          std::memcpy(buffer.get() + position, nan_literal, nan_len);
+          position += nan_len;
+        } else {
+          constexpr char inf_literal[] = "Infinity";
+          constexpr size_t inf_len = sizeof(inf_literal) - 1;
+          if (v < 0) {
+            buffer.get()[position] = '-';
+            ++position;
+          }
+          std::memcpy(buffer.get() + position, inf_literal, inf_len);
+          position += inf_len;
+        }
+        return;
+      }
+#endif
+
       // We could specialize for float.
       char *end = simdjson::internal::to_chars(buffer.get() + position, nullptr,
                                                double(v));
@@ -47940,7 +58219,9 @@ simdjson_inline void string_builder::escape_and_append_with_quotes() noexcept {
 #endif

 simdjson_inline void string_builder::append_raw(const char *c) noexcept {
-  size_t len = std::strlen(c);
+  // char_traits::length is constexpr; lets the compiler fold the length
+  // when called with a pointer to a compile-time-constant string.
+  size_t len = std::char_traits<char>::length(c);
   append_raw(c, len);
 }

@@ -47959,6 +58240,14 @@ simdjson_inline void string_builder::append_raw(const char *str,
     position += len;
   }
 }
+
+template <size_t N>
+simdjson_inline void string_builder::append_raw_n(const char *str) noexcept {
+  if (capacity_check(N)) {
+    std::memcpy(buffer.get() + position, str, N);
+    position += N;
+  }
+}
 #if SIMDJSON_SUPPORTS_CONCEPTS
 // Support for optional types (std::optional, etc.)
 template <concepts::optional_type T>
@@ -47988,7 +58277,7 @@ simdjson_inline void string_builder::append(const T &value) {
 #if SIMDJSON_SUPPORTS_RANGES && SIMDJSON_SUPPORTS_CONCEPTS
 // Support for range-based appending (std::ranges::view, etc.)
 template <std::ranges::range R>
-  requires(!std::is_convertible<R, std::string_view>::value && !require_custom_serialization<R>)
+  requires(!std::is_convertible<R, std::string_view>::value && !concepts::optional_type<R> && !require_custom_serialization<R>)
 simdjson_inline void string_builder::append(const R &range) noexcept {
   auto it = std::ranges::begin(range);
   auto end = std::ranges::end(range);
@@ -48302,16 +58591,6 @@ simdjson_inline int count_ones(uint64_t input_num) {
 }
 #endif

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
-                                         uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
-  *result = value1 + value2;
-  return *result < value1;
-#else
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-#endif
-}

 } // unnamed namespace
 } // namespace ppc64
@@ -49196,7 +59475,7 @@ public:
 #if SIMDJSON_SUPPORTS_RANGES && SIMDJSON_SUPPORTS_CONCEPTS
   // Support for range-based appending (std::ranges::view, etc.)
   template <std::ranges::range R>
-requires (!std::is_convertible<R, std::string_view>::value && !require_custom_serialization<R>)
+requires (!std::is_convertible<R, std::string_view>::value && !concepts::optional_type<R> && !require_custom_serialization<R>)
   simdjson_inline void append(const R &range) noexcept;
 #endif
   /**
@@ -49210,6 +59489,15 @@ requires (!std::is_convertible<R, std::string_view>::value && !require_custom_se
    * There is no UTF-8 validation.
    */
   simdjson_inline void append_raw(const char *str, size_t len) noexcept;
+
+  /**
+   * Append exactly N characters from str. The length is a template parameter
+   * so the compiler can fully inline the memcpy with a compile-time-constant
+   * size, avoiding the libc call. Used for compile-time-constant keys in the
+   * reflection struct atom.
+   */
+  template <size_t N>
+  simdjson_inline void append_raw_n(const char *str) noexcept;
 #if SIMDJSON_EXCEPTIONS
   /**
    * Creates an std::string from the written JSON buffer.
@@ -49261,6 +59549,26 @@ requires (!std::is_convertible<R, std::string_view>::value && !require_custom_se
    */
   simdjson_inline size_t size() const noexcept;

+  // ============================================================
+  // Internal hooks for the position-as-local writer in json_builder.h.
+  // These exist so the reflection atom code can hold buffer pointer,
+  // position and capacity in registers across long write chains rather
+  // than reloading them after every char* write (strict aliasing
+  // forces those reloads when accessed via members of *this). User
+  // code should NOT call these directly.
+  // ============================================================
+  simdjson_inline char *unsafe_data() noexcept { return buffer.get(); }
+  simdjson_inline size_t unsafe_position() const noexcept { return position; }
+  simdjson_inline size_t unsafe_capacity() const noexcept { return capacity; }
+  simdjson_inline void unsafe_set_position(size_t p) noexcept { position = p; }
+  /// Make capacity available for at least `n` more bytes after the current
+  /// position. Returns false if the allocation failed.
+  simdjson_inline bool unsafe_grow(size_t needed_total_capacity) noexcept {
+    grow_buffer(needed_total_capacity);
+    return is_valid;
+  }
+  simdjson_inline bool unsafe_is_valid() const noexcept { return is_valid; }
+
 private:
   /**
    * Returns true if we can write at least upcoming_bytes bytes.
@@ -49274,7 +59582,7 @@ private:
    * If the allocation fails, is_valid is set to false. We expect
    * that this function would not be repeatedly called.
    */
-  simdjson_inline void grow_buffer(size_t desired_capacity);
+  inline void grow_buffer(size_t desired_capacity);

   /**
    * We use this helper function to make sure that is_valid is kept consistent.
@@ -49331,6 +59639,7 @@ simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initi
 /* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_STRING_BUILDER_H */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/builder/json_string_builder.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/concepts.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/annotations.h" */
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
 #if SIMDJSON_STATIC_REFLECTION

@@ -49348,64 +59657,370 @@ namespace simdjson {
 namespace ppc64 {
 namespace builder {

-template <class T>
+// Forward-declare helpers defined in json_string_builder-inl.h so the
+// writer-based atom code below can call them (the -inl.h is not yet
+// included at the point this header is parsed; without these forwards,
+// name lookup falls back to the wrong outer namespace).
+namespace internal {
+simdjson_really_inline char *write_uint_jeaiii(char *p, uint64_t v) noexcept;
+simdjson_inline char *write_double(char *p, double v) noexcept;
+} // namespace internal
+simdjson_really_inline size_t write_string_escaped(const std::string_view input, char *out);
+
+SIMDJSON_PUSH_DISABLE_WARNINGS
+SIMDJSON_DISABLE_GCC_WARNING(-Warray-bounds)
+#if !defined(__clang__)
+SIMDJSON_DISABLE_GCC_WARNING(-Wstringop-overflow)
+#endif
+
+// =============================================================
+// `writer`: position-as-local hot-path writer used by the reflection
+// atom code below. Holds the buffer pointer, write position and
+// capacity in three fields that, once `writer` itself is a stack-local
+// in the caller and all atom() functions are inlined, become true
+// register-resident locals after SROA. Glaze achieves the same effect
+// by passing `B&& b, auto&& ix` through every helper. Holding `pos`
+// in a register (rather than as a member of string_builder) is what
+// breaks the strict-aliasing penalty on every char* write through the
+// buffer, which forces a reload of `b.position` and `b.capacity`
+// after every byte.
+//
+// basic_writer<false> (below) is the unchecked variant: the caller has
+// already reserved enough capacity for everything the write chain can
+// produce (see bound_detail::size_bound), so ensure() compiles away.
+// =============================================================
+template <bool Checked>
+struct basic_writer {
+  static constexpr bool checked = Checked;
+  char *ptr;        // buffer pointer (refreshed after a grow)
+  size_t pos;       // write position (local)
+  size_t cap;       // capacity (refreshed after a grow)
+  string_builder &sb;  // back-ref for grow / sync
+
+  // Snapshot string_builder state into a writer for the duration of
+  // a write chain.
+  simdjson_really_inline basic_writer(string_builder &builder) noexcept
+      : ptr(builder.unsafe_data())
+      , pos(builder.unsafe_position())
+      , cap(builder.unsafe_capacity())
+      , sb(builder) {}
+
+  // Write the local position back to the underlying string_builder.
+  // Caller is responsible for invoking before the writer is dropped
+  // (otherwise data is lost). Idempotent.
+  simdjson_really_inline void sync() noexcept {
+    sb.unsafe_set_position(pos);
+  }
+
+  // Ensure at least `n` more bytes of free capacity. Grows the
+  // underlying buffer if needed (rare path). Returns false on
+  // allocation failure.
+  simdjson_really_inline bool ensure(size_t n) noexcept {
+    // pos <= cap, and cap is the size of a live allocation, so pos + n
+    // cannot wrap when n is a small constant or a compile-time length.
+    // Callers passing a size derived from input (the string atoms) must
+    // bound it against pos themselves. Keep the `pos + n <= cap` form:
+    // `n <= cap - pos` is measurably slower once the serializer is inlined.
+    if (simdjson_likely(pos + n <= cap)) { return true; }
+    return grow_slow(n);
+  }
+
+  simdjson_never_inline bool grow_slow(size_t n) noexcept {
+    // Detect overflow.
+    // This is pedantic except maybe on 32-bit targets.
+    if (simdjson_unlikely(pos + n < pos)) return false;
+    sb.unsafe_set_position(pos);
+    // even if 2*capacity overflows, the (std::max) below will pick the needed value,
+    // so we do not need a separate overflow check here.
+    if (!sb.unsafe_grow((std::max)(cap * 2, pos + n))) {
+      // The string_builder freed its buffer and is now invalid (null buffer,
+      // zero capacity and position). Mirror that state so that every later
+      // ensure() fails too: callers only return from the current atom, and
+      // their callers keep writing.
+      ptr = nullptr;
+      pos = 0;
+      cap = 0;
+      return false;
+    }
+    ptr = sb.unsafe_data();
+    cap = sb.unsafe_capacity();
+    return true;
+  }
+};
+
+// The unchecked writer writes into a raw buffer that the caller sized with
+// serialized_size_bound: it never grows and needs no string_builder.
+template <>
+struct basic_writer<false> {
+  static constexpr bool checked = false;
+  char *ptr;
+  size_t pos;
+
+  simdjson_really_inline basic_writer(char *buffer, size_t position) noexcept
+      : ptr(buffer), pos(position) {}
+
+  simdjson_really_inline bool ensure(size_t) const noexcept { return true; }
+};
+
+using writer = basic_writer<true>;
+using unchecked_writer = basic_writer<false>;
+
+// Bytes reserved past the size bound for an unchecked writer: it may then
+// write a little past the end of what it produces (e.g., copy keys as whole
+// 16-byte blocks).
+inline constexpr size_t unchecked_slack = 64;
+
+consteval size_t padded_key_length(size_t length) {
+  return (length + 15) / 16 * 16;
+}
+
+// === Helper: invoke a string_builder member that writes variable-length
+// content (escape_and_append_with_quotes etc), syncing the writer's local
+// state before the call and reloading after. Used for string fields where
+// rewriting the entire SIMD escape path through the writer would be a much
+// bigger refactor. f may be user code (a with<Adapter> serializer) that
+// throws: the exception then propagates to the caller.
+template <class W, class F>
+simdjson_really_inline void call_through_string_builder(W &w, F &&f) noexcept(noexcept(f(w.sb))) {
+  w.sync();
+  f(w.sb);
+  w.ptr = w.sb.unsafe_data();
+  w.pos = w.sb.unsafe_position();
+  w.cap = w.sb.unsafe_capacity();
+}
+
+// Helpers implementing the serialization side of the annotations (see
+// simdjson/annotations.h) for reflected structures.
+namespace annotation_detail {
+
+// A member is serialized unless it is annotated with skip or skip_serializing.
+consteval bool is_serialized_member(std::meta::info dm) {
+  return !simdjson::detail::has_annotation(dm, ^^simdjson::detail::skip_tag)
+      && !simdjson::detail::has_annotation(dm, ^^simdjson::detail::skip_serializing_tag);
+}
+
+// False when the member has a skip_serializing_if<pred> annotation and
+// pred(value) is true.
+template <auto dm, typename V>
+simdjson_really_inline bool should_serialize(const V &value) {
+  constexpr std::meta::info skip_if_type = simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::skip_serializing_if_t);
+  if constexpr (skip_if_type != std::meta::info{}) {
+    using skip_if = typename [: skip_if_type :];
+    return !skip_if::predicate(value);
+  } else {
+    (void)value;
+    return true;
+  }
+}
+
+// Serialize a member value, through its with<Adapter> annotation when the
+// adapter provides a serialize function.
+template <auto dm, class W, typename V>
+simdjson_really_inline void atom_member(W &w, const V &value) {
+  constexpr std::meta::info with_type = simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::with_t);
+  if constexpr (with_type != std::meta::info{}) {
+    using adapter = typename [: with_type :]::adapter;
+    if constexpr (requires(string_builder &b) { adapter::serialize(b, value); }) {
+      call_through_string_builder(w, [&](string_builder &b) { adapter::serialize(b, value); });
+    } else {
+      atom(w, value);
+    }
+  } else {
+    atom(w, value);
+  }
+}
+
+// Write the "key":value pairs of the members of t (without the braces), each
+// preceded by a comma unless it is the first one. The members of a member
+// annotated with flatten are written in its place.
+template <class W, class T>
+simdjson_really_inline void atom_fields(W &w, const T &t, bool &first) {
+  // Per-field block: ensure key+value worst case, then write key + value
+  // through the writer's local pos. For arithmetic fields, the integer
+  // write happens directly via write_uint_jeaiii on w.ptr+w.pos, so pos
+  // never round-trips through memory.
+  template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+    if constexpr (is_serialized_member(dm)) {
+      if (should_serialize<dm>(t.[:dm:])) {
+        if constexpr (simdjson::detail::has_annotation(dm, ^^simdjson::detail::flatten_tag)) {
+          static_assert(std::meta::is_class_type(simdjson::detail::flattened_type(dm)));
+          using flattened = std::remove_cvref_t<decltype(t.[:dm:])>;
+          static_assert(!concepts::container_but_not_string<flattened> && !concepts::string_view_keyed_map<flattened> &&
+                        !concepts::appendable_containers<flattened> && !concepts::optional_type<flattened> &&
+                        !concepts::smart_pointer<flattened> && !std::is_same_v<flattened, std::string> &&
+                        !std::is_same_v<flattened, std::string_view> && !require_custom_serialization<flattened>,
+                        "simdjson::flatten requires a member whose type is a structure serialized member by member");
+          atom_fields(w, t.[:dm:], first);
+        } else {
+          // Copy the key as whole 16-byte blocks from a zero-padded copy (one
+          // load and one store); ensure() reserves the padded length, and the
+          // unchecked writer has slack past its bound. Prior related work:
+          // jsonifier copies a power-of-two padded key and advances the cursor
+          // by the real length (serialize_impl.hpp, packed_blitter,
+          // https://github.com/nihilai-collective/Jsonifier).
+          constexpr const char* key_name = simdjson::get_json_key_name<dm>();
+          constexpr size_t first_key_len = constevalutil::consteval_to_quoted_escaped(key_name).size() + 1;
+          constexpr size_t rest_key_len = first_key_len + 1;
+          constexpr auto first_key = std::define_static_string(
+              constevalutil::consteval_to_quoted_escaped(key_name) + ":" +
+              std::string(padded_key_length(first_key_len) - first_key_len, '\0'));
+          constexpr auto rest_key = std::define_static_string(
+              std::string(",") + constevalutil::consteval_to_quoted_escaped(key_name) + ":" +
+              std::string(padded_key_length(rest_key_len) - rest_key_len, '\0'));
+          if (!w.ensure(padded_key_length(rest_key_len))) { return; }
+          if (first) {
+            std::memcpy(w.ptr + w.pos, first_key, padded_key_length(first_key_len));
+            w.pos += first_key_len;
+          } else {
+            std::memcpy(w.ptr + w.pos, rest_key, padded_key_length(rest_key_len));
+            w.pos += rest_key_len;
+          }
+          first = false;
+          atom_member<dm>(w, t.[:dm:]);
+        }
+      }
+    }
+  };
+}
+
+} // namespace annotation_detail
+
+template <class W, class T>
   requires(concepts::container_but_not_string<T> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &t) {
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
   auto it = t.begin();
   auto end = t.end();
   if (it == end) {
-    b.append_raw("[]");
+    if (!w.ensure(2)) return;
+    std::memcpy(w.ptr + w.pos, "[]", 2);
+    w.pos += 2;
     return;
   }
-  b.append('[');
-  atom(b, *it);
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = '[';
+  atom(w, *it);
   ++it;
   for (; it != end; ++it) {
-    b.append(',');
-    atom(b, *it);
+    if (!w.ensure(1)) return;
+    w.ptr[w.pos++] = ',';
+    atom(w, *it);
   }
-  b.append(']');
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = ']';
 }

-template <class T>
+template <class W, class T>
   requires(std::is_same_v<T, std::string> ||
            std::is_same_v<T, std::string_view> ||
            std::is_same_v<T, const char *> ||
            std::is_same_v<T, char>)
-constexpr void atom(string_builder &b, const T &t) {
-  b.escape_and_append_with_quotes(t);
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
+  // Inline the escape path through the writer so we never round-trip
+  // pos through memory for string fields (Twitter is dominated by
+  // these -- sync/reload around each string was a real cost).
+  std::string_view input;
+  if constexpr (std::is_same_v<T, char>) {
+    input = std::string_view(&t, 1);
+  } else {
+    input = std::string_view(t);
+  }
+  // Worst-case escape: every byte expands to \uXXXX (6 chars), plus 2 quotes.
+  // Guard against w.pos + 2 + 6 * input.size() wrapping for huge inputs -- if
+  // it wrapped to a small value, ensure() would spuriously succeed and the
+  // subsequent escape would overflow the buffer. max - w.pos cannot wrap, and
+  // size < (max - pos) / 6 implies pos + 6 * size + 6 <= max.
+  // Note that this is pedantic except maybe on 32-bit targets.
+  if constexpr (W::checked) {
+    if (simdjson_unlikely(input.size() >= ((std::numeric_limits<size_t>::max)() - w.pos) / 6)) { return; }
+    if (!w.ensure(2 + 6 * input.size())) { return; }
+  }
+  w.ptr[w.pos++] = '"';
+  w.pos += write_string_escaped(input, w.ptr + w.pos);
+  w.ptr[w.pos++] = '"';
 }

-template <concepts::string_view_keyed_map T>
+template <class W, concepts::string_view_keyed_map T>
   requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &m) {
+simdjson_really_inline constexpr void atom(W &w, const T &m) {
   if (m.empty()) {
-    b.append_raw("{}");
+    if (!w.ensure(2)) return;
+    std::memcpy(w.ptr + w.pos, "{}", 2);
+    w.pos += 2;
     return;
   }
-  b.append('{');
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = '{';
   bool first = true;
   for (const auto& [key, value] : m) {
     if (!first) {
-      b.append(',');
+      if (!w.ensure(1)) return;
+      w.ptr[w.pos++] = ',';
     }
     first = false;
-    // Keys must be convertible to string_view per the concept
-    b.escape_and_append_with_quotes(key);
-    b.append(':');
-    atom(b, value);
+    // Keys must be convertible to string_view per the concept.
+    std::string_view key_sv(key);
+    // Guard against w.pos + 3 + 6 * key_sv.size() wrapping for huge keys, if
+    // it wrapped to a small value, ensure() would spuriously succeed and the
+    // subsequent escape would overflow the buffer. max - w.pos cannot wrap, and
+    // size < (max - pos) / 6 implies pos + 6 * size + 6 <= max.
+    // Note that this is pedantic except maybe on 32-bit targets.
+    if constexpr (W::checked) {
+      if (simdjson_unlikely(key_sv.size() >= ((std::numeric_limits<size_t>::max)() - w.pos) / 6)) { return; }
+      if (!w.ensure(2 + 6 * key_sv.size() + 1)) { return; }
+    }
+    w.ptr[w.pos++] = '"';
+    w.pos += write_string_escaped(key_sv, w.ptr + w.pos);
+    w.ptr[w.pos++] = '"';
+    w.ptr[w.pos++] = ':';
+    atom(w, value);
   }
-  b.append('}');
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = '}';
 }


-template<typename number_type,
+template<class W, typename number_type,
          typename = typename std::enable_if<std::is_arithmetic<number_type>::value && !std::is_same_v<number_type, char>>::type>
-constexpr void atom(string_builder &b, const number_type t) {
-  b.append(t);
+simdjson_really_inline constexpr void atom(W &w, const number_type t) {
+  // Booleans / floats: defer to string_builder (rare path; keeps writer hot
+  // path free of float-formatter machinery). For integers, write directly
+  // via jeaiii using local pos.
+  if constexpr (std::is_same_v<number_type, bool>) {
+    if (t) {
+      if (!w.ensure(4)) return;
+      std::memcpy(w.ptr + w.pos, "true", 4);
+      w.pos += 4;
+    } else {
+      if (!w.ensure(5)) return;
+      std::memcpy(w.ptr + w.pos, "false", 5);
+      w.pos += 5;
+    }
+  } else if constexpr (std::is_floating_point_v<number_type>) {
+    if constexpr (W::checked) {
+      call_through_string_builder(w, [&](string_builder &b) { b.append(t); });
+    } else {
+      w.pos = size_t(internal::write_double(w.ptr + w.pos, double(t)) - w.ptr);
+    }
+  } else if constexpr (std::is_unsigned_v<number_type>) {
+    if (!w.ensure(20)) return;
+    char *end = internal::write_uint_jeaiii(
+        w.ptr + w.pos, static_cast<uint64_t>(t));
+    w.pos = static_cast<size_t>(end - w.ptr);
+  } else {
+    // signed integral
+    if (!w.ensure(20)) return;
+    using U = typename std::make_unsigned<number_type>::type;
+    bool negative = t < 0;
+    U pv = negative ? U(0) - static_cast<U>(t) : static_cast<U>(t);
+    w.ptr[w.pos] = '-';
+    w.pos += negative;
+    char *end = internal::write_uint_jeaiii(
+        w.ptr + w.pos, static_cast<uint64_t>(pv));
+    w.pos = static_cast<size_t>(end - w.ptr);
+  }
 }

-template <class T>
+template <class W, class T>
   requires(std::is_class_v<T> && !concepts::container_but_not_string<T> &&
            !concepts::string_view_keyed_map<T> &&
            !concepts::optional_type<T> &&
@@ -49415,92 +60030,259 @@ template <class T>
            !std::is_same_v<T, std::string_view> &&
            !std::is_same_v<T, const char*> &&
            !std::is_same_v<T, char> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &t) {
-  int i = 0;
-  b.append('{');
-  template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
-    if (i != 0)
-      b.append(',');
-    constexpr auto key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(dm)));
-    b.append_raw(key);
-    b.append(':');
-    atom(b, t.[:dm:]);
-    i++;
-  };
-  b.append('}');
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
+  if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag)) {
+    // A transparent structure is serialized as its single member.
+    constexpr auto dm = simdjson::detail::transparent_member(^^T);
+    annotation_detail::atom_member<dm>(w, t.[:dm:]);
+  } else {
+    bool first = true;
+    if (!w.ensure(1)) { return; }
+    w.ptr[w.pos++] = '{';
+    annotation_detail::atom_fields(w, t, first);
+    if (!w.ensure(1)) { return; }
+    w.ptr[w.pos++] = '}';
+  }
 }

 // Support for optional types (std::optional, etc.)
-template <concepts::optional_type T>
+template <class W, concepts::optional_type T>
   requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &opt) {
+simdjson_really_inline constexpr void atom(W &w, const T &opt) {
   if (opt) {
-    atom(b, opt.value());
+    atom(w, opt.value());
   } else {
-    b.append_raw("null");
+    if (!w.ensure(4)) return;
+    std::memcpy(w.ptr + w.pos, "null", 4);
+    w.pos += 4;
   }
 }

 // Support for smart pointers (std::unique_ptr, std::shared_ptr, etc.)
-template <concepts::smart_pointer T>
+template <class W, concepts::smart_pointer T>
   requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &ptr) {
+simdjson_really_inline constexpr void atom(W &w, const T &ptr) {
   if (ptr) {
-    atom(b, *ptr);
+    atom(w, *ptr);
   } else {
-    b.append_raw("null");
+    if (!w.ensure(4)) return;
+    std::memcpy(w.ptr + w.pos, "null", 4);
+    w.pos += 4;
   }
 }

 // Support for enums - serialize as string representation using expand approach from P2996R12
-template <typename T>
+template <class W, typename T>
   requires(std::is_enum_v<T> && !require_custom_serialization<T>)
-void atom(string_builder &b, const T &e) {
+simdjson_really_inline void atom(W &w, const T &e) {
 #if SIMDJSON_STATIC_REFLECTION
-  constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
+  static constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
   template for (constexpr auto enum_val : enumerators) {
-    constexpr auto enum_str = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(enum_val)));
+    constexpr auto enum_str = std::define_static_string(constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<enum_val>()));
+    constexpr size_t enum_str_len = std::char_traits<char>::length(enum_str);
     if (e == [:enum_val:]) {
-      b.append_raw(enum_str);
+      if (!w.ensure(enum_str_len)) return;
+      std::memcpy(w.ptr + w.pos, enum_str, enum_str_len);
+      w.pos += enum_str_len;
       return;
     }
   };
   // Fallback to integer if enum value not found
-  atom(b, static_cast<std::underlying_type_t<T>>(e));
+  atom(w, static_cast<std::underlying_type_t<T>>(e));
 #else
   // Fallback: serialize as integer if reflection not available
-  atom(b, static_cast<std::underlying_type_t<T>>(e));
+  atom(w, static_cast<std::underlying_type_t<T>>(e));
 #endif
 }

 // Support for appendable containers that don't have operator[] (sets, etc.)
-template <concepts::appendable_containers T>
+template <class W, concepts::appendable_containers T>
   requires(!concepts::container_but_not_string<T> && !concepts::string_view_keyed_map<T> &&
            !concepts::optional_type<T> && !concepts::smart_pointer<T> &&
            !std::is_same_v<T, std::string> &&
            !std::is_same_v<T, std::string_view> && !std::is_same_v<T, const char*> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &container) {
+simdjson_really_inline constexpr void atom(W &w, const T &container) {
   if (container.empty()) {
-    b.append_raw("[]");
+    if (!w.ensure(2)) return;
+    std::memcpy(w.ptr + w.pos, "[]", 2);
+    w.pos += 2;
     return;
   }
-  b.append('[');
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = '[';
   bool first = true;
   for (const auto& item : container) {
     if (!first) {
-      b.append(',');
+      if (!w.ensure(1)) return;
+      w.ptr[w.pos++] = ',';
     }
     first = false;
-    atom(b, item);
+    atom(w, item);
+  }
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = ']';
+}
+
+// =============================================================
+// Size bound: an upper bound on the number of bytes that atom(w, t) writes.
+// Computing it first lets append() reserve the capacity once and then run
+// the whole write chain through an unchecked_writer, without a capacity
+// check before every write. It mirrors the atom() overloads above.
+//
+// Prior related work: jsonifier sizes the document first, counting 6 bytes
+// per string byte, resizes once to that bound plus slack, and writes with
+// no capacity check on each store. to_json() does that resize through
+// std::string::resize_and_overwrite. See serializer.hpp and
+// serialize_impl.hpp in https://github.com/nihilai-collective/Jsonifier
+// and https://nihilai-collective.net/serialization.
+// =============================================================
+namespace bound_detail {
+
+// Whether size_bound covers everything that atom() writes for T: not when a
+// member is serialized by a with<Adapter> serializer, which writes an unknown
+// amount through the string_builder.
+template <class T>
+consteval bool is_bounded() {
+  if constexpr (require_custom_serialization<T>) {
+    return false;
+  } else if constexpr (std::is_same_v<T, std::string> || std::is_same_v<T, std::string_view> ||
+                       std::is_same_v<T, const char *> || std::is_arithmetic_v<T> || std::is_enum_v<T>) {
+    return true;
+  } else if constexpr (concepts::optional_type<T> || concepts::smart_pointer<T>) {
+    return is_bounded<std::remove_cvref_t<decltype(*std::declval<const T &>())>>();
+  } else if constexpr (concepts::string_view_keyed_map<T>) {
+    return is_bounded<std::remove_cvref_t<typename T::mapped_type>>();
+  } else if constexpr (concepts::container_but_not_string<T> || concepts::appendable_containers<T>) {
+    return is_bounded<std::remove_cvref_t<std::ranges::range_value_t<T>>>();
+  } else {
+    bool bounded = true;
+    template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+      if constexpr (annotation_detail::is_serialized_member(dm)) {
+        bounded = bounded && simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::with_t) == std::meta::info{} &&
+                  is_bounded<std::remove_cvref_t<decltype(std::declval<const T &>().[:dm:])>>();
+      }
+    };
+    return bounded;
+  }
+}
+
+template <class T>
+consteval size_t enum_bound() {
+  size_t bound = 20; // the integer fallback
+  template for (constexpr auto enum_val : std::define_static_array(std::meta::enumerators_of(^^T))) {
+    constexpr size_t len = std::char_traits<char>::length(std::define_static_string(
+        constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<enum_val>())));
+    bound = (std::max)(bound, len);
+  };
+  return bound;
+}
+
+template <class T>
+simdjson_really_inline size_t size_bound(const T &t) noexcept;
+
+// Bound for the "key":value pairs of a structure, commas included.
+template <class T>
+simdjson_really_inline size_t fields_bound(const T &t) noexcept {
+  size_t bound = 0;
+  template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+    if constexpr (annotation_detail::is_serialized_member(dm)) {
+      if constexpr (simdjson::detail::has_annotation(dm, ^^simdjson::detail::flatten_tag)) {
+        bound += fields_bound(t.[:dm:]);
+      } else {
+        constexpr size_t rest_key_len = constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<dm>()).size() + 2;
+        bound += rest_key_len + size_bound(t.[:dm:]);
+      }
+    }
+  };
+  return bound;
+}
+
+template <class T>
+simdjson_really_inline size_t size_bound([[maybe_unused]] const T &t) noexcept {
+  if constexpr (std::is_same_v<T, char>) {
+    return 2 + 6;
+  } else if constexpr (std::is_same_v<T, std::string> || std::is_same_v<T, std::string_view> ||
+                       std::is_same_v<T, const char *>) {
+    // Every byte may become \uXXXX, plus the quotes.
+    return 2 + 6 * std::string_view(t).size();
+  } else if constexpr (std::is_same_v<T, bool>) {
+    return 5;
+  } else if constexpr (std::is_floating_point_v<T>) {
+    return simdjson::internal::to_chars_buffer_size;
+  } else if constexpr (std::is_arithmetic_v<T>) {
+    return 20;
+  } else if constexpr (std::is_enum_v<T>) {
+    return enum_bound<T>();
+  } else if constexpr (concepts::optional_type<T> || concepts::smart_pointer<T>) {
+    return t ? size_bound(*t) : 4;
+  } else if constexpr (concepts::string_view_keyed_map<T>) {
+    size_t bound = 2;
+    for (const auto &[key, value] : t) {
+      // comma, quotes, colon
+      bound += 4 + 6 * std::string_view(key).size() + size_bound(value);
+    }
+    return bound;
+  } else if constexpr (concepts::container_but_not_string<T> || concepts::appendable_containers<T>) {
+    using value_type = std::remove_cvref_t<std::ranges::range_value_t<T>>;
+    if constexpr (std::is_arithmetic_v<value_type> && !std::is_same_v<value_type, char>) {
+      // A fixed bound per element: no need to visit them.
+      return 2 + size_t(std::ranges::distance(t)) * (1 + size_bound(value_type{}));
+    } else {
+      size_t bound = 2;
+#if defined(__GNUC__) && !defined(__clang__)
+#pragma GCC novector // a vector loop is slower on short containers
+#endif
+#pragma GCC unroll 4
+      for (const auto &item : t) {
+        bound += 1 + size_bound(item);
+      }
+      return bound;
+    }
+  } else if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag)) {
+    constexpr auto dm = simdjson::detail::transparent_member(^^T);
+    return size_bound(t.[:dm:]);
+  } else {
+    return 2 + fields_bound(t);
+  }
+}
+
+} // namespace bound_detail
+
+// Write t through an unchecked writer when its size bound is available,
+// reserving that many bytes first, and through the checked writer otherwise.
+template <class T>
+simdjson_really_inline void append_bounded(string_builder &b, const T &t) {
+  // On 32-bit systems, the bound could overflow: keep the checked writer.
+  if constexpr (sizeof(size_t) >= 8 && bound_detail::is_bounded<T>()) {
+    const size_t bound = bound_detail::size_bound(t) + unchecked_slack;
+    const size_t pos = b.unsafe_position();
+    // The bound is a sum of in-memory sizes times a small constant: it cannot
+    // overflow on a 64-bit system. Be pedantic elsewhere.
+    if (sizeof(size_t) >= 8 || bound <= (std::numeric_limits<size_t>::max)() - pos) {
+      const size_t cap = b.unsafe_capacity();
+      // Grow geometrically so that many small appends stay amortized.
+      if (pos + bound <= cap || b.unsafe_grow((std::max)(cap * 2, pos + bound))) {
+        unchecked_writer w(b.unsafe_data(), pos);
+        atom(w, t);
+        b.unsafe_set_position(w.pos);
+      }
+      return;
+    }
   }
-  b.append(']');
+  writer w(b);
+  atom(w, t);
+  w.sync();
 }

-// append functions that delegate to atom functions for primitive types
+// append() -- top-level entry. Each overload constructs a stack-local
+// writer, runs atom(w, t) through the inlined call chain, then syncs
+// the local position back into the string_builder.
 template <class T>
   requires(std::is_arithmetic_v<T> && !std::is_same_v<T, char>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  writer w(b);
+  atom(w, t);
+  w.sync();
 }

 template <class T>
@@ -49508,20 +60290,22 @@ template <class T>
            std::is_same_v<T, std::string_view> ||
            std::is_same_v<T, const char *> ||
            std::is_same_v<T, char>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  writer w(b);
+  atom(w, t);
+  w.sync();
 }

 template <concepts::optional_type T>
   requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  append_bounded(b, t);
 }

 template <concepts::smart_pointer T>
   requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  append_bounded(b, t);
 }

 template <concepts::appendable_containers T>
@@ -49529,14 +60313,14 @@ template <concepts::appendable_containers T>
            !concepts::optional_type<T> && !concepts::smart_pointer<T> &&
            !std::is_same_v<T, std::string> &&
            !std::is_same_v<T, std::string_view> && !std::is_same_v<T, const char*> && !require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  append_bounded(b, t);
 }

 template <concepts::string_view_keyed_map T>
   requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  append_bounded(b, t);
 }

 // works for struct
@@ -49550,39 +60334,15 @@ template <class Z>
            !std::is_same_v<Z, std::string_view> &&
            !std::is_same_v<Z, const char*> &&
            !std::is_same_v<Z, char> && !require_custom_serialization<Z>)
-void append(string_builder &b, const Z &z) {
-  int i = 0;
-  b.append('{');
-  template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^Z, std::meta::access_context::unchecked()))) {
-    if (i != 0)
-      b.append(',');
-    constexpr auto key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(dm)));
-    b.append_raw(key);
-    b.append(':');
-    atom(b, z.[:dm:]);
-    i++;
-  };
-  b.append('}');
+simdjson_inline void append(string_builder &b, const Z &z) {
+  append_bounded(b, z);
 }

 // works for container that have begin() and end() iterators
 template <class Z>
   requires(concepts::container_but_not_string<Z> && !require_custom_serialization<Z>)
-void append(string_builder &b, const Z &z) {
-  auto it = z.begin();
-  auto end = z.end();
-  if (it == end) {
-    b.append_raw("[]");
-    return;
-  }
-  b.append('[');
-  atom(b, *it);
-  ++it;
-  for (; it != end; ++it) {
-    b.append(',');
-    atom(b, *it);
-  }
-  b.append(']');
+simdjson_inline void append(string_builder &b, const Z &z) {
+  append_bounded(b, z);
 }

 template <class Z>
@@ -49593,22 +60353,40 @@ void append(string_builder &b, const Z &z) {


 template <class Z>
-simdjson_warn_unused simdjson_result<std::string> to_json_string(const Z &z, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
-  string_builder b(initial_capacity);
-  append(b, z);
-  std::string_view s;
-  if(auto e = b.view().get(s); e) { return e; }
-  return std::string(s);
+simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
+  if constexpr (sizeof(size_t) >= 8 && bound_detail::is_bounded<Z>()) {
+    // Write straight into s, sized by the bound: no intermediate buffer, no copy.
+    // Prior related work: jsonifier's serializeJson resizes once through
+    // resize_and_overwrite (serializer.hpp).
+    (void)initial_capacity;
+    const size_t bound = bound_detail::size_bound(z) + unchecked_slack;
+    auto write = [&z](char *p) noexcept {
+      unchecked_writer w(p, 0);
+      atom(w, z);
+      return w.pos;
+    };
+#if defined(__cpp_lib_string_resize_and_overwrite) && __cpp_lib_string_resize_and_overwrite >= 202110L
+    s.resize_and_overwrite(bound, [&write](char *p, size_t) noexcept { return write(p); });
+#else
+    s.resize(bound);
+    s.resize(write(s.data()));
+#endif
+    return SUCCESS;
+  } else {
+    string_builder b(initial_capacity);
+    append(b, z);
+    std::string_view view;
+    if(auto e = b.view().get(view); e) { return e; }
+    s.assign(view);
+    return SUCCESS;
+  }
 }

 template <class Z>
-simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
-  string_builder b(initial_capacity);
-  append(b, z);
-  std::string_view view;
-  if(auto e = b.view().get(view); e) { return e; }
-  s.assign(view);
-  return SUCCESS;
+simdjson_warn_unused simdjson_result<std::string> to_json_string(const Z &z, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
+  std::string s;
+  if(auto e = to_json(z, s, initial_capacity); e) { return e; }
+  return s;
 }

 template <class Z>
@@ -49621,40 +60399,41 @@ string_builder& operator<<(string_builder& b, const Z& z) {
 template<constevalutil::fixed_string... FieldNames, typename T>
   requires(std::is_class_v<T> && (sizeof...(FieldNames) > 0))
 void extract_from(string_builder &b, const T &obj) {
-  // Helper to check if a field name matches any of the requested fields
-  auto should_extract = [](std::string_view field_name) constexpr -> bool {
-    return ((FieldNames.view() == field_name) || ...);
-  };
-
-  b.append('{');
+  writer w(b);
+  if (!w.ensure(1)) { w.sync(); return; }
+  w.ptr[w.pos++] = '{';
   bool first = true;
-
   // Iterate through all members of T using reflection
-  template for (constexpr auto mem : std::define_static_array(
-      std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
-
+  static constexpr auto members = std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()));
+  template for (constexpr auto mem : members) {
     if constexpr (std::meta::is_public(mem)) {
-      constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
+      static constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));

       // Only serialize this field if it's in our list of requested fields
-      if constexpr (should_extract(key)) {
-        if (!first) {
-          b.append(',');
+      if constexpr (((FieldNames.view() == key) || ...)) {
+        static constexpr auto first_key = std::define_static_string(
+            constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)) + ":");
+        static constexpr auto rest_key = std::define_static_string(
+            std::string(",") + constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)) + ":");
+        constexpr size_t first_key_len = std::char_traits<char>::length(first_key);
+        constexpr size_t rest_key_len = std::char_traits<char>::length(rest_key);
+        if (!w.ensure(rest_key_len)) { w.sync(); return; }
+        if (first) {
+          std::memcpy(w.ptr + w.pos, first_key, first_key_len);
+          w.pos += first_key_len;
+        } else {
+          std::memcpy(w.ptr + w.pos, rest_key, rest_key_len);
+          w.pos += rest_key_len;
         }
         first = false;
-
-        // Serialize the key
-        constexpr auto quoted_key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)));
-        b.append_raw(quoted_key);
-        b.append(':');
-
-        // Serialize the value
-        atom(b, obj.[:mem:]);
+        atom(w, obj.[:mem:]);
       }
     }
   };

-  b.append('}');
+  if (!w.ensure(1)) { w.sync(); return; }
+  w.ptr[w.pos++] = '}';
+  w.sync();
 }

 template<constevalutil::fixed_string... FieldNames, typename T>
@@ -49667,25 +60446,18 @@ simdjson_warn_unused simdjson_result<std::string> extract_from(const T &obj, siz
   return std::string(s);
 }

+SIMDJSON_POP_DISABLE_WARNINGS
+
 } // namespace builder
 } // namespace ppc64
 // Alias the function template to 'to' in the global namespace
 template <class Z>
 simdjson_warn_unused simdjson_result<std::string> to_json(const Z &z, size_t initial_capacity = ppc64::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
-  ppc64::builder::string_builder b(initial_capacity);
-  ppc64::builder::append(b, z);
-  std::string_view s;
-  if(auto e = b.view().get(s); e) { return e; }
-  return std::string(s);
+  return ppc64::builder::to_json_string(z, initial_capacity);
 }
 template <class Z>
 simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = ppc64::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
-  ppc64::builder::string_builder b(initial_capacity);
-  ppc64::builder::append(b, z);
-  std::string_view view;
-  if(auto e = b.view().get(view); e) { return e; }
-  s.assign(view);
-  return SUCCESS;
+  return ppc64::builder::to_json(z, s, initial_capacity);
 }
 // Global namespace function for extract_from
 template<constevalutil::fixed_string... FieldNames, typename T>
@@ -49831,6 +60603,7 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
 /* including simdjson/generic/builder/json_string_builder-inl.h for ppc64: #include "simdjson/generic/builder/json_string_builder-inl.h" */
 /* begin file simdjson/generic/builder/json_string_builder-inl.h for ppc64 */
 #include <array>
+#include <cmath>
 #include <cstring>
 #include <limits>
 #include <type_traits>
@@ -49865,6 +60638,11 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
 #define SIMDJSON_EXPERIMENTAL_HAS_LSX 1
 #endif
 #endif
+#if defined(__loongarch_asx)
+#ifndef SIMDJSON_EXPERIMENTAL_HAS_LASX
+#define SIMDJSON_EXPERIMENTAL_HAS_LASX 1
+#endif
+#endif
 #if defined(__riscv_v_intrinsic) && __riscv_v_intrinsic >= 11000 &&            \
     defined(__riscv_vector)
 #ifndef SIMDJSON_EXPERIMENTAL_HAS_RVV
@@ -49884,6 +60662,9 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
 #endif
 #if SIMDJSON_EXPERIMENTAL_HAS_SSE2
 #include <emmintrin.h>
+#if defined(__AVX2__)
+#include <immintrin.h>
+#endif
 #ifdef _MSC_VER
 #include <intrin.h>
 #endif
@@ -49891,6 +60672,9 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
 #if SIMDJSON_EXPERIMENTAL_HAS_LSX
 #include <lsxintrin.h>
 #endif
+#if SIMDJSON_EXPERIMENTAL_HAS_LASX
+#include <lasxintrin.h>
+#endif
 #if SIMDJSON_EXPERIMENTAL_HAS_RVV
 #include <riscv_vector.h>
 #endif
@@ -49940,105 +60724,6 @@ inline bool has_json_escapable_byte(uint64_t x) {

 **/

-SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline bool
-simple_needs_escaping(std::string_view v) {
-  for (char c : v) {
-    // a table lookup is faster than a series of comparisons
-    if (json_quotable_character[static_cast<uint8_t>(c)]) {
-      return true;
-    }
-  }
-  return false;
-}
-
-#if SIMDJSON_EXPERIMENTAL_HAS_NEON
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
-  if (view.size() < 16) {
-    return simple_needs_escaping(view);
-  }
-  size_t i = 0;
-  uint8x16_t running = vdupq_n_u8(0);
-  uint8x16_t v34 = vdupq_n_u8(34);
-  uint8x16_t v92 = vdupq_n_u8(92);
-
-  for (; i + 15 < view.size(); i += 16) {
-    uint8x16_t word = vld1q_u8((const uint8_t *)view.data() + i);
-    running = vorrq_u8(running, vceqq_u8(word, v34));
-    running = vorrq_u8(running, vceqq_u8(word, v92));
-    running = vorrq_u8(running, vcltq_u8(word, vdupq_n_u8(32)));
-  }
-  if (i < view.size()) {
-    uint8x16_t word =
-        vld1q_u8((const uint8_t *)view.data() + view.length() - 16);
-    running = vorrq_u8(running, vceqq_u8(word, v34));
-    running = vorrq_u8(running, vceqq_u8(word, v92));
-    running = vorrq_u8(running, vcltq_u8(word, vdupq_n_u8(32)));
-  }
-  return vmaxvq_u32(vreinterpretq_u32_u8(running)) != 0;
-}
-#elif SIMDJSON_EXPERIMENTAL_HAS_SSE2
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
-  if (view.size() < 16) {
-    return simple_needs_escaping(view);
-  }
-  size_t i = 0;
-  __m128i running = _mm_setzero_si128();
-  for (; i + 15 < view.size(); i += 16) {
-
-    __m128i word =
-        _mm_loadu_si128(reinterpret_cast<const __m128i *>(view.data() + i));
-    running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(34)));
-    running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(92)));
-    running = _mm_or_si128(
-        running, _mm_cmpeq_epi8(_mm_subs_epu8(word, _mm_set1_epi8(31)),
-                                _mm_setzero_si128()));
-  }
-  if (i < view.size()) {
-    __m128i word = _mm_loadu_si128(
-        reinterpret_cast<const __m128i *>(view.data() + view.length() - 16));
-    running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(34)));
-    running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(92)));
-    running = _mm_or_si128(
-        running, _mm_cmpeq_epi8(_mm_subs_epu8(word, _mm_set1_epi8(31)),
-                                _mm_setzero_si128()));
-  }
-  return _mm_movemask_epi8(running) != 0;
-}
-#elif SIMDJSON_EXPERIMENTAL_HAS_PPC64
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
-  if (view.size() < 16) {
-    return simple_needs_escaping(view);
-  }
-  size_t i = 0;
-  __vector unsigned char running = vec_splats((unsigned char)0);
-  __vector unsigned char v34 = vec_splats((unsigned char)34);
-  __vector unsigned char v92 = vec_splats((unsigned char)92);
-  __vector unsigned char v32 = vec_splats((unsigned char)32);
-
-  for (; i + 15 < view.size(); i += 16) {
-    __vector unsigned char word =
-        vec_vsx_ld(0, reinterpret_cast<const unsigned char *>(view.data() + i));
-    running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v34));
-    running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v92));
-    running = vec_or(running,
-        (__vector unsigned char)vec_cmplt(word, v32));
-  }
-  if (i < view.size()) {
-    __vector unsigned char word = vec_vsx_ld(
-        0, reinterpret_cast<const unsigned char *>(view.data() + view.length() - 16));
-    running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v34));
-    running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v92));
-    running = vec_or(running,
-        (__vector unsigned char)vec_cmplt(word, v32));
-  }
-  return !vec_all_eq(running, vec_splats((unsigned char)0));
-}
-#else
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
-  return simple_needs_escaping(view);
-}
-#endif
-
 // Scalar fallback for finding next quotable character
 SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline size_t
 find_next_json_quotable_character_scalar(const std::string_view view,
@@ -50129,6 +60814,51 @@ find_next_json_quotable_character(const std::string_view view,
   size_t current = len - remaining;
   return find_next_json_quotable_character_scalar(view, current);
 }
+#elif SIMDJSON_EXPERIMENTAL_HAS_LASX
+simdjson_inline size_t
+find_next_json_quotable_character(const std::string_view view,
+                                  size_t location) noexcept {
+  const size_t len = view.size();
+  const uint8_t *ptr =
+      reinterpret_cast<const uint8_t *>(view.data()) + location;
+  size_t remaining = len - location;
+
+  // SIMD constants for characters requiring escape
+  __m256i v34 = __lasx_xvreplgr2vr_b(34);  // '"'
+  __m256i v92 = __lasx_xvreplgr2vr_b(92);  // '\\'
+  __m256i v32 = __lasx_xvreplgr2vr_b(32);  // control char threshold
+
+  while (remaining >= 32) {
+    __m256i word = __lasx_xvld(ptr, 0);
+
+    // Check for quotable characters: '"', '\\', or control chars (< 32)
+    __m256i needs_escape = __lasx_xvseq_b(word, v34);
+    needs_escape = __lasx_xvor_v(needs_escape, __lasx_xvseq_b(word, v92));
+    needs_escape = __lasx_xvor_v(needs_escape, __lasx_xvslt_bu(word, v32));
+
+    if (!__lasx_xbz_v(needs_escape)) {
+      // Found a quotable character - locate it via the four 64-bit lanes
+      uint64_t lane0 = __lasx_xvpickve2gr_du(needs_escape, 0);
+      uint64_t lane1 = __lasx_xvpickve2gr_du(needs_escape, 1);
+      uint64_t lane2 = __lasx_xvpickve2gr_du(needs_escape, 2);
+      uint64_t lane3 = __lasx_xvpickve2gr_du(needs_escape, 3);
+      size_t offset = ptr - reinterpret_cast<const uint8_t *>(view.data());
+      if (lane0 != 0) {
+        return offset + trailing_zeroes(lane0) / 8;
+      } else if (lane1 != 0) {
+        return offset + 8 + trailing_zeroes(lane1) / 8;
+      } else if (lane2 != 0) {
+        return offset + 16 + trailing_zeroes(lane2) / 8;
+      } else {
+        return offset + 24 + trailing_zeroes(lane3) / 8;
+      }
+    }
+    ptr += 32;
+    remaining -= 32;
+  }
+  size_t current = len - remaining;
+  return find_next_json_quotable_character_scalar(view, current);
+}
 #elif SIMDJSON_EXPERIMENTAL_HAS_LSX
 simdjson_inline size_t
 find_next_json_quotable_character(const std::string_view view,
@@ -50284,6 +61014,254 @@ SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline void escape_json_char(char c, char *&o
   }
 }

+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2 || SIMDJSON_EXPERIMENTAL_HAS_NEON
+#define SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE 1
+#endif
+
+#if SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2
+
+using escape_vector = __m128i;
+// Mask bits that each input byte contributes to escape_bitmask().
+static constexpr unsigned escape_mask_bits = 1;
+
+simdjson_inline escape_vector escape_load16(const uint8_t *p) noexcept {
+  return _mm_loadu_si128(reinterpret_cast<const __m128i *>(p));
+}
+
+simdjson_inline void escape_store16(char *out, escape_vector v) noexcept {
+  _mm_storeu_si128(reinterpret_cast<__m128i *>(out), v);
+}
+
+// Builds a vector whose bytes 0..7 come from a and bytes 8..15 from b.
+simdjson_inline escape_vector escape_load8x2(const uint8_t *a,
+                                             const uint8_t *b) noexcept {
+  return _mm_unpacklo_epi64(
+      _mm_loadl_epi64(reinterpret_cast<const __m128i *>(a)),
+      _mm_loadl_epi64(reinterpret_cast<const __m128i *>(b)));
+}
+
+// Builds a vector whose bytes 0..3 come from a and bytes 4..7 from b. The
+// upper half is unspecified; callers only look at the low eight bits.
+simdjson_inline escape_vector escape_load4x2(const uint8_t *a,
+                                             const uint8_t *b) noexcept {
+  int32_t a32, b32;
+  memcpy(&a32, a, 4);
+  memcpy(&b32, b, 4);
+  return _mm_unpacklo_epi32(_mm_cvtsi32_si128(a32), _mm_cvtsi32_si128(b32));
+}
+
+// Sets every bit of byte k when byte k of v is a quotable character ('"',
+// '\\' or a control character).
+simdjson_inline escape_vector escape_flags(escape_vector v) noexcept {
+  const __m128i v34 = _mm_set1_epi8(34); // '"'
+  const __m128i v92 = _mm_set1_epi8(92); // '\\'
+  const __m128i v31 = _mm_set1_epi8(31); // for control char detection
+  __m128i needs_escape = _mm_cmpeq_epi8(v, v34);
+  needs_escape = _mm_or_si128(needs_escape, _mm_cmpeq_epi8(v, v92));
+  return _mm_or_si128(
+      needs_escape, _mm_cmpeq_epi8(_mm_subs_epu8(v, v31), _mm_setzero_si128()));
+}
+
+// True when any byte needs escaping. Kept separate from escape_bitmask
+// because some instruction sets can answer it without leaving the vector
+// register file; on SSE2 the compiler folds the two together.
+simdjson_inline bool escape_any(escape_vector flags) noexcept {
+  return _mm_movemask_epi8(flags) != 0;
+}
+
+// A 16-bit mask with bit k set when byte k needed escaping.
+simdjson_inline uint64_t escape_bitmask(escape_vector flags) noexcept {
+  return uint64_t(uint32_t(_mm_movemask_epi8(flags)));
+}
+
+#else // SIMDJSON_EXPERIMENTAL_HAS_NEON
+
+using escape_vector = uint8x16_t;
+static constexpr unsigned escape_mask_bits = 4;
+
+simdjson_inline escape_vector escape_load16(const uint8_t *p) noexcept {
+  return vld1q_u8(p);
+}
+
+simdjson_inline void escape_store16(char *out, escape_vector v) noexcept {
+  vst1q_u8(reinterpret_cast<uint8_t *>(out), v);
+}
+
+simdjson_inline escape_vector escape_load8x2(const uint8_t *a,
+                                             const uint8_t *b) noexcept {
+  uint64_t a64, b64;
+  memcpy(&a64, a, 8);
+  memcpy(&b64, b, 8);
+  return vreinterpretq_u8_u64(vsetq_lane_u64(b64, vdupq_n_u64(a64), 1));
+}
+
+simdjson_inline escape_vector escape_load4x2(const uint8_t *a,
+                                             const uint8_t *b) noexcept {
+  uint32_t a32, b32;
+  memcpy(&a32, a, 4);
+  memcpy(&b32, b, 4);
+  return vreinterpretq_u8_u32(vsetq_lane_u32(b32, vdupq_n_u32(a32), 1));
+}
+
+simdjson_inline escape_vector escape_flags(escape_vector v) noexcept {
+  uint8x16_t needs_escape = vceqq_u8(v, vdupq_n_u8(34));              // '"'
+  needs_escape = vorrq_u8(needs_escape, vceqq_u8(v, vdupq_n_u8(92))); // '\\'
+  return vorrq_u8(needs_escape, vcltq_u8(v, vdupq_n_u8(32)));
+}
+
+simdjson_inline bool escape_any(escape_vector flags) noexcept {
+  return vmaxvq_u32(vreinterpretq_u32_u8(flags)) != 0;
+}
+
+// Four bits per byte rather than one: escape_block and the tail paths scale
+// their shifts by escape_mask_bits to match.
+simdjson_inline uint64_t escape_bitmask(escape_vector flags) noexcept {
+  uint8x8_t narrowed = vshrn_n_u16(vreinterpretq_u16_u8(flags), 4);
+  return vget_lane_u64(vreinterpret_u64_u8(narrowed), 0);
+}
+
+#endif // instruction set selection
+
+// The escape bitmask of a 16-byte block.
+simdjson_inline uint64_t escape_mask(escape_vector v) noexcept {
+  return escape_bitmask(escape_flags(v));
+}
+
+// Copies n bytes with n < 16, using overlapping loads and stores. It never
+// reads more than n bytes from src, nor writes more than n bytes to dst.
+simdjson_inline void copy_lt16(char *dst, const uint8_t *src,
+                               size_t n) noexcept {
+  if (n >= 8) {
+    memcpy(dst, src, 8);
+    memcpy(dst + n - 8, src + n - 8, 8);
+  } else if (n >= 4) {
+    memcpy(dst, src, 4);
+    memcpy(dst + n - 4, src + n - 4, 4);
+  } else if (n > 0) {
+    dst[0] = char(src[0]);
+    dst[n >> 1] = char(src[n >> 1]);
+    dst[n - 1] = char(src[n - 1]);
+  }
+}
+
+// Escapes the bytes of src in the range [i, blockend), given that m is the
+// (non-zero) escape mask for that range: the escape_mask_bits-wide lane of m
+// at byte k is non-zero when src[i + k] requires escaping. Returns the updated
+// output pointer.
+simdjson_never_inline char *escape_block(const uint8_t *src, char *out,
+                                         size_t i, size_t blockend,
+                                         uint64_t m) noexcept {
+  constexpr uint64_t lane = (uint64_t(1) << escape_mask_bits) - 1;
+
+  size_t pos = i; // first byte not yet copied
+  while (m) {
+    const size_t tz = trailing_zeroes(m);
+    const size_t next = i + tz / escape_mask_bits;
+    // Copy the run of safe bytes that precedes this escape.
+    copy_lt16(out, src + pos, next - pos);
+    out += next - pos;
+    escape_json_char(char(src[next]), out);
+    pos = next + 1;
+    m &= ~(lane << tz);
+  }
+  // Copy whatever follows the last escape.
+  copy_lt16(out, src + pos, blockend - pos);
+  return out + (blockend - pos);
+}
+
+// Writes the escaped version of input to out, returning the number of bytes
+// written.
+simdjson_really_inline size_t write_string_escaped(const std::string_view input, char *out) {
+  const size_t len = input.size();
+  const uint8_t *src = reinterpret_cast<const uint8_t *>(input.data());
+  const char *const initout = out;
+
+  size_t i = 0;
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2 && defined(__AVX2__)
+  while (i + 32 <= len) {
+    const __m256i word = _mm256_loadu_si256(reinterpret_cast<const __m256i *>(src + i));
+    const __m256i flags = _mm256_or_si256(
+        _mm256_or_si256(_mm256_cmpeq_epi8(word, _mm256_set1_epi8(34)),   // '"'
+                        _mm256_cmpeq_epi8(word, _mm256_set1_epi8(92))),  // '\\'
+        _mm256_cmpeq_epi8(_mm256_subs_epu8(word, _mm256_set1_epi8(31)),
+                          _mm256_setzero_si256()));                      // control
+    const uint32_t mask = uint32_t(_mm256_movemask_epi8(flags));
+    if (simdjson_likely(mask == 0)) {
+      _mm256_storeu_si256(reinterpret_cast<__m256i *>(out), word);
+      out += 32;
+    } else {
+      for (size_t half = 0; half < 32; half += 16) {
+        const uint64_t m = (mask >> half) & 0xFFFF;
+        if (m == 0) {
+          escape_store16(out, escape_load16(src + i + half));
+          out += 16;
+        } else {
+          out = escape_block(src, out, i + half, i + half + 16, m);
+        }
+      }
+    }
+    i += 32;
+  }
+#endif
+  while (i + 16 <= len) {
+    escape_vector word = escape_load16(src + i);
+    escape_vector flags = escape_flags(word);
+    if (simdjson_likely(!escape_any(flags))) {
+      escape_store16(out, word);
+      out += 16;
+    } else {
+      out = escape_block(src, out, i, i + 16, escape_bitmask(flags));
+    }
+    i += 16;
+  }
+  if (i < len) {
+    const size_t rem = len - i;
+    uint64_t m;
+    if (len >= 16) {
+      // The last 16 bytes of the input are in bounds. Bit k of that block's
+      // mask belongs to input position len - 16 + k, so shift it down to align
+      // bit 0 with position i.
+      m = escape_mask(escape_load16(src + len - 16)) >>
+          (escape_mask_bits * (16 - rem));
+    } else if (len >= 8) {
+      // Here i == 0 and rem == len. Two overlapping 8-byte loads cover
+      // [0, 8) and [len - 8, len), which is the whole input since len < 16.
+      uint64_t mm = escape_mask(escape_load8x2(src, src + len - 8));
+      constexpr uint64_t low8 = (uint64_t(1) << (escape_mask_bits * 8)) - 1;
+      m = (mm & low8) |
+          ((mm >> (escape_mask_bits * 8)) << (escape_mask_bits * (len - 8)));
+    } else if (len >= 4) {
+      // Same idea with two overlapping 4-byte loads.
+      uint64_t mm = escape_mask(escape_load4x2(src, src + len - 4));
+      constexpr uint64_t low4 = (uint64_t(1) << (escape_mask_bits * 4)) - 1;
+      m = (mm & low4) | (((mm >> (escape_mask_bits * 4)) & low4)
+                         << (escape_mask_bits * (len - 4)));
+    } else {
+      // Fewer than 4 bytes: at most three table lookups, no need for SIMD.
+      for (size_t k = 0; k < len; k++) {
+        uint8_t c = src[k];
+        if (json_quotable_character[c]) {
+          escape_json_char(char(c), out);
+        } else {
+          *out++ = char(c);
+        }
+      }
+      return size_t(out - initout);
+    }
+    if (m == 0) {
+      copy_lt16(out, src + i, rem);
+      out += rem;
+    } else {
+      out = escape_block(src, out, i, len, m);
+    }
+  }
+  return size_t(out - initout);
+}
+
+#else // SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
 // Writes the escaped version of input to out, returning the number of bytes
 // written. Uses SIMD position finding to locate quotable characters efficiently.
 inline size_t write_string_escaped(const std::string_view input, char *out) {
@@ -50313,9 +61291,13 @@ inline size_t write_string_escaped(const std::string_view input, char *out) {
     escape_json_char(input[location], out);
     location += 1;
   }
-  return out - initout;
+  return size_t(out - initout);
 }

+#endif // SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+#undef SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
+
 simdjson_inline string_builder::string_builder(size_t initial_capacity)
     : buffer(new(std::nothrow) char[initial_capacity]), position(0),
       capacity(buffer.get() != nullptr ? initial_capacity : 0),
@@ -50338,7 +61320,7 @@ simdjson_inline bool string_builder::capacity_check(size_t upcoming_bytes) {
   return is_valid;
 }

-simdjson_inline void string_builder::grow_buffer(size_t desired_capacity) {
+inline void string_builder::grow_buffer(size_t desired_capacity) {
   if (!is_valid) {
     return;
   }
@@ -50394,81 +61376,136 @@ simdjson_inline void string_builder::clear() noexcept {

 namespace internal {

-template <typename number_type, typename = typename std::enable_if<
-                                    std::is_unsigned<number_type>::value>::type>
-simdjson_really_inline int int_log2(number_type x) {
-  return 63 - leading_zeroes(uint64_t(x) | 1);
-}
-
-simdjson_really_inline int fast_digit_count_32(uint32_t x) {
-  static uint64_t table[] = {
-      4294967296,  8589934582,  8589934582,  8589934582,  12884901788,
-      12884901788, 12884901788, 17179868184, 17179868184, 17179868184,
-      21474826480, 21474826480, 21474826480, 21474826480, 25769703776,
-      25769703776, 25769703776, 30063771072, 30063771072, 30063771072,
-      34349738368, 34349738368, 34349738368, 34349738368, 38554705664,
-      38554705664, 38554705664, 41949672960, 41949672960, 41949672960,
-      42949672960, 42949672960};
-  return uint32_t((x + table[int_log2(x)]) >> 32);
-}
-
-simdjson_really_inline int fast_digit_count_64(uint64_t x) {
-  static uint64_t table[] = {9,
-                             99,
-                             999,
-                             9999,
-                             99999,
-                             999999,
-                             9999999,
-                             99999999,
-                             999999999,
-                             9999999999,
-                             99999999999,
-                             999999999999,
-                             9999999999999,
-                             99999999999999,
-                             999999999999999ULL,
-                             9999999999999999ULL,
-                             99999999999999999ULL,
-                             999999999999999999ULL,
-                             9999999999999999999ULL};
-  int y = (19 * int_log2(x) >> 6);
-  y += x > table[y];
-  return y + 1;
-}
-
-template <typename number_type, typename = typename std::enable_if<
-                                    std::is_unsigned<number_type>::value>::type>
-simdjson_really_inline size_t digit_count(number_type v) noexcept {
-  static_assert(sizeof(number_type) == 8 || sizeof(number_type) == 4 ||
-                    sizeof(number_type) == 2 || sizeof(number_type) == 1,
-                "We only support 8-bit, 16-bit, 32-bit and 64-bit numbers");
-  SIMDJSON_IF_CONSTEXPR(sizeof(number_type) <= 4) {
-    return fast_digit_count_32(static_cast<uint32_t>(v));
+// Integer to decimal: James Edward Anhalt III's algorithm
+static const char jeaiii_dd[201] =
+    "00010203040506070809101112131415161718192021222324252627282930313233343536373839"
+    "40414243444546474849505152535455565758596061626364656667686970717273747576777879"
+    "8081828384858687888990919293949596979899";
+static const char jeaiii_fd[201] =
+    "0\0" "1\0" "2\0" "3\0" "4\0" "5\0" "6\0" "7\0" "8\0" "9\0"
+    "10111213141516171819202122232425262728293031323334353637383940414243444546474849"
+    "50515253545556575859606162636465666768697071727374757677787980818283848586878889"
+    "90919293949596979899";
+
+simdjson_really_inline void jeaiii_write_dd(char *p, uint64_t k) noexcept {
+  std::memcpy(p, &jeaiii_dd[2 * k], 2);
+}
+simdjson_really_inline void jeaiii_write_fd(char *p, uint64_t k) noexcept {
+  std::memcpy(p, &jeaiii_fd[2 * k], 2);
+}
+
+// Caller guarantees n < 10^8. Writes 1 to 8 digits.
+simdjson_really_inline char *jeaiii_lt1e8(char *b, uint32_t n) noexcept {
+  constexpr uint64_t mask24 = (uint64_t(1) << 24) - 1;
+  constexpr uint64_t mask32 = (uint64_t(1) << 32) - 1;
+  if (n < 100) {
+    jeaiii_write_fd(b, n);
+    return n < 10 ? b + 1 : b + 2;
+  }
+  if (n < 1000000) {
+    if (n < 10000) {
+      const uint32_t f0 = uint32_t(10 * (1 << 24) / 1e3 + 1) * n;
+      jeaiii_write_fd(b, f0 >> 24);
+      b -= n < 1000;
+      const uint32_t f2 = uint32_t(f0 & mask24) * 100;
+      jeaiii_write_dd(b + 2, f2 >> 24);
+      return b + 4;
+    }
+    const uint64_t f0 = uint64_t(10 * (1ull << 32) / 1e5 + 1) * n;
+    jeaiii_write_fd(b, f0 >> 32);
+    b -= n < 100000;
+    const uint64_t f2 = (f0 & mask32) * 100;
+    jeaiii_write_dd(b + 2, f2 >> 32);
+    const uint64_t f4 = (f2 & mask32) * 100;
+    jeaiii_write_dd(b + 4, f4 >> 32);
+    return b + 6;
+  }
+  const uint64_t f0 = uint64_t(10 * (1ull << 48) / 1e7 + 1) * n >> 16;
+  jeaiii_write_fd(b, f0 >> 32);
+  b -= n < 10000000;
+  const uint64_t f2 = (f0 & mask32) * 100;
+  jeaiii_write_dd(b + 2, f2 >> 32);
+  const uint64_t f4 = (f2 & mask32) * 100;
+  jeaiii_write_dd(b + 4, f4 >> 32);
+  const uint64_t f6 = (f4 & mask32) * 100;
+  jeaiii_write_dd(b + 6, f6 >> 32);
+  return b + 8;
+}
+
+// Caller guarantees z < 10^8. Always writes exactly 8 digits.
+simdjson_really_inline char *jeaiii_8_digits(char *b, uint32_t z) noexcept {
+  constexpr uint64_t mask32 = (uint64_t(1) << 32) - 1;
+  const uint64_t f0 = (uint64_t((1ull << 48) / 1e6 + 1) * z >> 16) + 1;
+  jeaiii_write_dd(b, f0 >> 32);
+  const uint64_t f2 = (f0 & mask32) * 100;
+  jeaiii_write_dd(b + 2, f2 >> 32);
+  const uint64_t f4 = (f2 & mask32) * 100;
+  jeaiii_write_dd(b + 4, f4 >> 32);
+  const uint64_t f6 = (f4 & mask32) * 100;
+  jeaiii_write_dd(b + 6, f6 >> 32);
+  return b + 8;
+}
+
+// Caller guarantees 10^8 <= n < 2^32. Writes 9 or 10 digits.
+simdjson_really_inline char *jeaiii_9_or_10(char *b, uint64_t n) noexcept {
+  constexpr uint64_t mask57 = (uint64_t(1) << 57) - 1;
+  const uint64_t f0 = uint64_t(10 * (1ull << 57) / 1e9 + 1) * n;
+  jeaiii_write_fd(b, f0 >> 57);
+  b -= n < 1000000000;
+  const uint64_t f2 = (f0 & mask57) * 100;
+  jeaiii_write_dd(b + 2, f2 >> 57);
+  const uint64_t f4 = (f2 & mask57) * 100;
+  jeaiii_write_dd(b + 4, f4 >> 57);
+  const uint64_t f6 = (f4 & mask57) * 100;
+  jeaiii_write_dd(b + 6, f6 >> 57);
+  const uint64_t f8 = (f6 & mask57) * 100;
+  jeaiii_write_dd(b + 8, f8 >> 57);
+  return b + 10;
+}
+
+simdjson_really_inline char *write_uint_jeaiii(char *b, uint64_t n) noexcept {
+  if (n < 100000000) {
+    return jeaiii_lt1e8(b, uint32_t(n));
+  }
+  if (n < (uint64_t(1) << 32)) {
+    return jeaiii_9_or_10(b, n);
+  }
+  // At least 10 digits: the low 8 digits, and 2 to 12 digits above them.
+  const uint32_t z = uint32_t(n % 100000000);
+  uint64_t u = n / 100000000;
+  if (u < 100000000) {
+    // u has 2 to 8 digits (if u < 10, n would be below 2^32).
+    b = jeaiii_lt1e8(b, uint32_t(u));
+  } else if (u < (uint64_t(1) << 32)) {
+    b = jeaiii_9_or_10(b, u);
+  } else {
+    // u has 11 or 12 digits: split off 8 more.
+    const uint32_t y = uint32_t(u % 100000000);
+    u /= 100000000;
+    b = jeaiii_lt1e8(b, uint32_t(u)); // 3 or 4 digits
+    b = jeaiii_8_digits(b, y);
   }
-  else {
-    return fast_digit_count_64(static_cast<uint64_t>(v));
-  }
-}
-static const char decimal_table[200] = {
-    0x30, 0x30, 0x30, 0x31, 0x30, 0x32, 0x30, 0x33, 0x30, 0x34, 0x30, 0x35,
-    0x30, 0x36, 0x30, 0x37, 0x30, 0x38, 0x30, 0x39, 0x31, 0x30, 0x31, 0x31,
-    0x31, 0x32, 0x31, 0x33, 0x31, 0x34, 0x31, 0x35, 0x31, 0x36, 0x31, 0x37,
-    0x31, 0x38, 0x31, 0x39, 0x32, 0x30, 0x32, 0x31, 0x32, 0x32, 0x32, 0x33,
-    0x32, 0x34, 0x32, 0x35, 0x32, 0x36, 0x32, 0x37, 0x32, 0x38, 0x32, 0x39,
-    0x33, 0x30, 0x33, 0x31, 0x33, 0x32, 0x33, 0x33, 0x33, 0x34, 0x33, 0x35,
-    0x33, 0x36, 0x33, 0x37, 0x33, 0x38, 0x33, 0x39, 0x34, 0x30, 0x34, 0x31,
-    0x34, 0x32, 0x34, 0x33, 0x34, 0x34, 0x34, 0x35, 0x34, 0x36, 0x34, 0x37,
-    0x34, 0x38, 0x34, 0x39, 0x35, 0x30, 0x35, 0x31, 0x35, 0x32, 0x35, 0x33,
-    0x35, 0x34, 0x35, 0x35, 0x35, 0x36, 0x35, 0x37, 0x35, 0x38, 0x35, 0x39,
-    0x36, 0x30, 0x36, 0x31, 0x36, 0x32, 0x36, 0x33, 0x36, 0x34, 0x36, 0x35,
-    0x36, 0x36, 0x36, 0x37, 0x36, 0x38, 0x36, 0x39, 0x37, 0x30, 0x37, 0x31,
-    0x37, 0x32, 0x37, 0x33, 0x37, 0x34, 0x37, 0x35, 0x37, 0x36, 0x37, 0x37,
-    0x37, 0x38, 0x37, 0x39, 0x38, 0x30, 0x38, 0x31, 0x38, 0x32, 0x38, 0x33,
-    0x38, 0x34, 0x38, 0x35, 0x38, 0x36, 0x38, 0x37, 0x38, 0x38, 0x38, 0x39,
-    0x39, 0x30, 0x39, 0x31, 0x39, 0x32, 0x39, 0x33, 0x39, 0x34, 0x39, 0x35,
-    0x39, 0x36, 0x39, 0x37, 0x39, 0x38, 0x39, 0x39,
-};
+  return jeaiii_8_digits(b, z);
+}
+
+// Writes v at p, which must have to_chars_buffer_size bytes available, and
+// returns the end of what was written.
+simdjson_inline char *write_double(char *p, double v) noexcept {
+#if SIMDJSON_ENABLE_NAN_INF
+  if (simdjson_unlikely(!std::isfinite(v))) {
+    if (std::isnan(v)) {
+      std::memcpy(p, "NaN", 3);
+      return p + 3;
+    }
+    if (v < 0) {
+      *p++ = '-';
+    }
+    std::memcpy(p, "Infinity", 8);
+    return p + 8;
+  }
+#endif
+  return simdjson::internal::to_chars(p, nullptr, v);
+}
 } // namespace internal

 template <typename number_type, typename>
@@ -50496,87 +61533,62 @@ simdjson_inline void string_builder::append(number_type v) noexcept {
     }
   }
   else SIMDJSON_IF_CONSTEXPR(std::is_unsigned<number_type>::value) {
-    // Process 4 digits at a time instead of 2, reducing store operations
-    // and divisions by approximately half for large numbers.
     constexpr size_t max_number_size = 20;
     if (capacity_check(max_number_size)) {
       using unsigned_type = typename std::make_unsigned<number_type>::type;
-      unsigned_type pv = static_cast<unsigned_type>(v);
-      size_t dc = internal::digit_count(pv);
-      char *write_pointer = buffer.get() + position + dc - 1;
-
-      // Process 4 digits per iteration for large numbers
-      while (pv >= 10000) {
-        unsigned_type q = pv / 10000;
-        unsigned_type r = pv % 10000;
-        unsigned_type r_hi = r / 100;  // High 2 digits of remainder
-        unsigned_type r_lo = r % 100;  // Low 2 digits of remainder
-        // Write low 2 digits first (rightmost), then high 2 digits
-        memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
-        memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
-        write_pointer -= 4;
-        pv = q;
-      }
-
-      // Handle remaining 1-4 digits with original 2-digit loop
-      while (pv >= 100) {
-        memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
-        write_pointer -= 2;
-        pv /= 100;
-      }
-      if (pv >= 10) {
-        *write_pointer-- = char('0' + (pv % 10));
-        pv /= 10;
-      }
-      *write_pointer = char('0' + pv);
-      position += dc;
+      char* end = internal::write_uint_jeaiii(
+          buffer.get() + position,
+          static_cast<uint64_t>(static_cast<unsigned_type>(v)));
+      position = end - buffer.get();
     }
   }
   else SIMDJSON_IF_CONSTEXPR(std::is_integral<number_type>::value) {
-    // Same 4-digit batching as unsigned path for signed integers
+    // 19 digits (max abs value of int64_t) + optional minus sign.
     constexpr size_t max_number_size = 20;
     if (capacity_check(max_number_size)) {
       using unsigned_type = typename std::make_unsigned<number_type>::type;
       bool negative = v < 0;
-      unsigned_type pv = static_cast<unsigned_type>(v);
-      if (negative) {
-        pv = 0 - pv; // the 0 is for Microsoft
-      }
-      size_t dc = internal::digit_count(pv);
-      // by always writing the minus sign, we avoid the branch.
+      // 0 - pv (rather than -pv) avoids an MSVC unary-minus warning.
+      unsigned_type pv = negative
+          ? unsigned_type(0) - static_cast<unsigned_type>(v)
+          : static_cast<unsigned_type>(v);
+      // Branchless: always write '-', advance only if negative.
       buffer.get()[position] = '-';
-      position += negative ? 1 : 0;
-      char *write_pointer = buffer.get() + position + dc - 1;
-
-      // Process 4 digits per iteration for large numbers
-      while (pv >= 10000) {
-        unsigned_type q = pv / 10000;
-        unsigned_type r = pv % 10000;
-        unsigned_type r_hi = r / 100;
-        unsigned_type r_lo = r % 100;
-        memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
-        memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
-        write_pointer -= 4;
-        pv = q;
-      }
-
-      // Handle remaining 1-4 digits
-      while (pv >= 100) {
-        memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
-        write_pointer -= 2;
-        pv /= 100;
-      }
-      if (pv >= 10) {
-        *write_pointer-- = char('0' + (pv % 10));
-        pv /= 10;
-      }
-      *write_pointer = char('0' + pv);
-      position += dc;
+      position += negative;
+      char* end = internal::write_uint_jeaiii(
+          buffer.get() + position, static_cast<uint64_t>(pv));
+      position = end - buffer.get();
     }
   }
   else SIMDJSON_IF_CONSTEXPR(std::is_floating_point<number_type>::value) {
-    constexpr size_t max_number_size = 24;
+    // Must reserve to_chars_buffer_size (40): only ~24 chars are emitted,
+    // but to_chars over-writes with fixed-size 16/17-byte copies so the
+    // compiler can inline mem* (see simdjson::internal::to_chars_buffer_size).
+    constexpr size_t max_number_size = simdjson::internal::to_chars_buffer_size;
     if (capacity_check(max_number_size)) {
+#if SIMDJSON_ENABLE_NAN_INF
+      // Check if the input might be NaN or infinity
+      if (simdjson_unlikely(!std::isfinite(v))) {
+        if (std::isnan(v)) {
+          constexpr char nan_literal[] = "NaN";
+          constexpr size_t nan_len = sizeof(nan_literal) - 1;
+
+          std::memcpy(buffer.get() + position, nan_literal, nan_len);
+          position += nan_len;
+        } else {
+          constexpr char inf_literal[] = "Infinity";
+          constexpr size_t inf_len = sizeof(inf_literal) - 1;
+          if (v < 0) {
+            buffer.get()[position] = '-';
+            ++position;
+          }
+          std::memcpy(buffer.get() + position, inf_literal, inf_len);
+          position += inf_len;
+        }
+        return;
+      }
+#endif
+
       // We could specialize for float.
       char *end = simdjson::internal::to_chars(buffer.get() + position, nullptr,
                                                double(v));
@@ -50637,7 +61649,9 @@ simdjson_inline void string_builder::escape_and_append_with_quotes() noexcept {
 #endif

 simdjson_inline void string_builder::append_raw(const char *c) noexcept {
-  size_t len = std::strlen(c);
+  // char_traits::length is constexpr; lets the compiler fold the length
+  // when called with a pointer to a compile-time-constant string.
+  size_t len = std::char_traits<char>::length(c);
   append_raw(c, len);
 }

@@ -50656,6 +61670,14 @@ simdjson_inline void string_builder::append_raw(const char *str,
     position += len;
   }
 }
+
+template <size_t N>
+simdjson_inline void string_builder::append_raw_n(const char *str) noexcept {
+  if (capacity_check(N)) {
+    std::memcpy(buffer.get() + position, str, N);
+    position += N;
+  }
+}
 #if SIMDJSON_SUPPORTS_CONCEPTS
 // Support for optional types (std::optional, etc.)
 template <concepts::optional_type T>
@@ -50685,7 +61707,7 @@ simdjson_inline void string_builder::append(const T &value) {
 #if SIMDJSON_SUPPORTS_RANGES && SIMDJSON_SUPPORTS_CONCEPTS
 // Support for range-based appending (std::ranges::view, etc.)
 template <std::ranges::range R>
-  requires(!std::is_convertible<R, std::string_view>::value && !require_custom_serialization<R>)
+  requires(!std::is_convertible<R, std::string_view>::value && !concepts::optional_type<R> && !require_custom_serialization<R>)
 simdjson_inline void string_builder::append(const R &range) noexcept {
   auto it = std::ranges::begin(range);
   auto end = std::ranges::end(range);
@@ -51013,16 +62035,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
 }
 #endif

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
-                                uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
-  return _addcarry_u64(0, value1, value2,
-                       reinterpret_cast<unsigned __int64 *>(result));
-#else
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-#endif
-}

 } // unnamed namespace
 } // namespace westmere
@@ -51601,16 +62613,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
 }
 #endif

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
-                                uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
-  return _addcarry_u64(0, value1, value2,
-                       reinterpret_cast<unsigned __int64 *>(result));
-#else
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-#endif
-}

 } // unnamed namespace
 } // namespace westmere
@@ -52210,7 +63212,7 @@ public:
 #if SIMDJSON_SUPPORTS_RANGES && SIMDJSON_SUPPORTS_CONCEPTS
   // Support for range-based appending (std::ranges::view, etc.)
   template <std::ranges::range R>
-requires (!std::is_convertible<R, std::string_view>::value && !require_custom_serialization<R>)
+requires (!std::is_convertible<R, std::string_view>::value && !concepts::optional_type<R> && !require_custom_serialization<R>)
   simdjson_inline void append(const R &range) noexcept;
 #endif
   /**
@@ -52224,6 +63226,15 @@ requires (!std::is_convertible<R, std::string_view>::value && !require_custom_se
    * There is no UTF-8 validation.
    */
   simdjson_inline void append_raw(const char *str, size_t len) noexcept;
+
+  /**
+   * Append exactly N characters from str. The length is a template parameter
+   * so the compiler can fully inline the memcpy with a compile-time-constant
+   * size, avoiding the libc call. Used for compile-time-constant keys in the
+   * reflection struct atom.
+   */
+  template <size_t N>
+  simdjson_inline void append_raw_n(const char *str) noexcept;
 #if SIMDJSON_EXCEPTIONS
   /**
    * Creates an std::string from the written JSON buffer.
@@ -52275,6 +63286,26 @@ requires (!std::is_convertible<R, std::string_view>::value && !require_custom_se
    */
   simdjson_inline size_t size() const noexcept;

+  // ============================================================
+  // Internal hooks for the position-as-local writer in json_builder.h.
+  // These exist so the reflection atom code can hold buffer pointer,
+  // position and capacity in registers across long write chains rather
+  // than reloading them after every char* write (strict aliasing
+  // forces those reloads when accessed via members of *this). User
+  // code should NOT call these directly.
+  // ============================================================
+  simdjson_inline char *unsafe_data() noexcept { return buffer.get(); }
+  simdjson_inline size_t unsafe_position() const noexcept { return position; }
+  simdjson_inline size_t unsafe_capacity() const noexcept { return capacity; }
+  simdjson_inline void unsafe_set_position(size_t p) noexcept { position = p; }
+  /// Make capacity available for at least `n` more bytes after the current
+  /// position. Returns false if the allocation failed.
+  simdjson_inline bool unsafe_grow(size_t needed_total_capacity) noexcept {
+    grow_buffer(needed_total_capacity);
+    return is_valid;
+  }
+  simdjson_inline bool unsafe_is_valid() const noexcept { return is_valid; }
+
 private:
   /**
    * Returns true if we can write at least upcoming_bytes bytes.
@@ -52288,7 +63319,7 @@ private:
    * If the allocation fails, is_valid is set to false. We expect
    * that this function would not be repeatedly called.
    */
-  simdjson_inline void grow_buffer(size_t desired_capacity);
+  inline void grow_buffer(size_t desired_capacity);

   /**
    * We use this helper function to make sure that is_valid is kept consistent.
@@ -52345,6 +63376,7 @@ simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initi
 /* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_STRING_BUILDER_H */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/builder/json_string_builder.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/concepts.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/annotations.h" */
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
 #if SIMDJSON_STATIC_REFLECTION

@@ -52362,64 +63394,370 @@ namespace simdjson {
 namespace westmere {
 namespace builder {

-template <class T>
+// Forward-declare helpers defined in json_string_builder-inl.h so the
+// writer-based atom code below can call them (the -inl.h is not yet
+// included at the point this header is parsed; without these forwards,
+// name lookup falls back to the wrong outer namespace).
+namespace internal {
+simdjson_really_inline char *write_uint_jeaiii(char *p, uint64_t v) noexcept;
+simdjson_inline char *write_double(char *p, double v) noexcept;
+} // namespace internal
+simdjson_really_inline size_t write_string_escaped(const std::string_view input, char *out);
+
+SIMDJSON_PUSH_DISABLE_WARNINGS
+SIMDJSON_DISABLE_GCC_WARNING(-Warray-bounds)
+#if !defined(__clang__)
+SIMDJSON_DISABLE_GCC_WARNING(-Wstringop-overflow)
+#endif
+
+// =============================================================
+// `writer`: position-as-local hot-path writer used by the reflection
+// atom code below. Holds the buffer pointer, write position and
+// capacity in three fields that, once `writer` itself is a stack-local
+// in the caller and all atom() functions are inlined, become true
+// register-resident locals after SROA. Glaze achieves the same effect
+// by passing `B&& b, auto&& ix` through every helper. Holding `pos`
+// in a register (rather than as a member of string_builder) is what
+// breaks the strict-aliasing penalty on every char* write through the
+// buffer, which forces a reload of `b.position` and `b.capacity`
+// after every byte.
+//
+// basic_writer<false> (below) is the unchecked variant: the caller has
+// already reserved enough capacity for everything the write chain can
+// produce (see bound_detail::size_bound), so ensure() compiles away.
+// =============================================================
+template <bool Checked>
+struct basic_writer {
+  static constexpr bool checked = Checked;
+  char *ptr;        // buffer pointer (refreshed after a grow)
+  size_t pos;       // write position (local)
+  size_t cap;       // capacity (refreshed after a grow)
+  string_builder &sb;  // back-ref for grow / sync
+
+  // Snapshot string_builder state into a writer for the duration of
+  // a write chain.
+  simdjson_really_inline basic_writer(string_builder &builder) noexcept
+      : ptr(builder.unsafe_data())
+      , pos(builder.unsafe_position())
+      , cap(builder.unsafe_capacity())
+      , sb(builder) {}
+
+  // Write the local position back to the underlying string_builder.
+  // Caller is responsible for invoking before the writer is dropped
+  // (otherwise data is lost). Idempotent.
+  simdjson_really_inline void sync() noexcept {
+    sb.unsafe_set_position(pos);
+  }
+
+  // Ensure at least `n` more bytes of free capacity. Grows the
+  // underlying buffer if needed (rare path). Returns false on
+  // allocation failure.
+  simdjson_really_inline bool ensure(size_t n) noexcept {
+    // pos <= cap, and cap is the size of a live allocation, so pos + n
+    // cannot wrap when n is a small constant or a compile-time length.
+    // Callers passing a size derived from input (the string atoms) must
+    // bound it against pos themselves. Keep the `pos + n <= cap` form:
+    // `n <= cap - pos` is measurably slower once the serializer is inlined.
+    if (simdjson_likely(pos + n <= cap)) { return true; }
+    return grow_slow(n);
+  }
+
+  simdjson_never_inline bool grow_slow(size_t n) noexcept {
+    // Detect overflow.
+    // This is pedantic except maybe on 32-bit targets.
+    if (simdjson_unlikely(pos + n < pos)) return false;
+    sb.unsafe_set_position(pos);
+    // even if 2*capacity overflows, the (std::max) below will pick the needed value,
+    // so we do not need a separate overflow check here.
+    if (!sb.unsafe_grow((std::max)(cap * 2, pos + n))) {
+      // The string_builder freed its buffer and is now invalid (null buffer,
+      // zero capacity and position). Mirror that state so that every later
+      // ensure() fails too: callers only return from the current atom, and
+      // their callers keep writing.
+      ptr = nullptr;
+      pos = 0;
+      cap = 0;
+      return false;
+    }
+    ptr = sb.unsafe_data();
+    cap = sb.unsafe_capacity();
+    return true;
+  }
+};
+
+// The unchecked writer writes into a raw buffer that the caller sized with
+// serialized_size_bound: it never grows and needs no string_builder.
+template <>
+struct basic_writer<false> {
+  static constexpr bool checked = false;
+  char *ptr;
+  size_t pos;
+
+  simdjson_really_inline basic_writer(char *buffer, size_t position) noexcept
+      : ptr(buffer), pos(position) {}
+
+  simdjson_really_inline bool ensure(size_t) const noexcept { return true; }
+};
+
+using writer = basic_writer<true>;
+using unchecked_writer = basic_writer<false>;
+
+// Bytes reserved past the size bound for an unchecked writer: it may then
+// write a little past the end of what it produces (e.g., copy keys as whole
+// 16-byte blocks).
+inline constexpr size_t unchecked_slack = 64;
+
+consteval size_t padded_key_length(size_t length) {
+  return (length + 15) / 16 * 16;
+}
+
+// === Helper: invoke a string_builder member that writes variable-length
+// content (escape_and_append_with_quotes etc), syncing the writer's local
+// state before the call and reloading after. Used for string fields where
+// rewriting the entire SIMD escape path through the writer would be a much
+// bigger refactor. f may be user code (a with<Adapter> serializer) that
+// throws: the exception then propagates to the caller.
+template <class W, class F>
+simdjson_really_inline void call_through_string_builder(W &w, F &&f) noexcept(noexcept(f(w.sb))) {
+  w.sync();
+  f(w.sb);
+  w.ptr = w.sb.unsafe_data();
+  w.pos = w.sb.unsafe_position();
+  w.cap = w.sb.unsafe_capacity();
+}
+
+// Helpers implementing the serialization side of the annotations (see
+// simdjson/annotations.h) for reflected structures.
+namespace annotation_detail {
+
+// A member is serialized unless it is annotated with skip or skip_serializing.
+consteval bool is_serialized_member(std::meta::info dm) {
+  return !simdjson::detail::has_annotation(dm, ^^simdjson::detail::skip_tag)
+      && !simdjson::detail::has_annotation(dm, ^^simdjson::detail::skip_serializing_tag);
+}
+
+// False when the member has a skip_serializing_if<pred> annotation and
+// pred(value) is true.
+template <auto dm, typename V>
+simdjson_really_inline bool should_serialize(const V &value) {
+  constexpr std::meta::info skip_if_type = simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::skip_serializing_if_t);
+  if constexpr (skip_if_type != std::meta::info{}) {
+    using skip_if = typename [: skip_if_type :];
+    return !skip_if::predicate(value);
+  } else {
+    (void)value;
+    return true;
+  }
+}
+
+// Serialize a member value, through its with<Adapter> annotation when the
+// adapter provides a serialize function.
+template <auto dm, class W, typename V>
+simdjson_really_inline void atom_member(W &w, const V &value) {
+  constexpr std::meta::info with_type = simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::with_t);
+  if constexpr (with_type != std::meta::info{}) {
+    using adapter = typename [: with_type :]::adapter;
+    if constexpr (requires(string_builder &b) { adapter::serialize(b, value); }) {
+      call_through_string_builder(w, [&](string_builder &b) { adapter::serialize(b, value); });
+    } else {
+      atom(w, value);
+    }
+  } else {
+    atom(w, value);
+  }
+}
+
+// Write the "key":value pairs of the members of t (without the braces), each
+// preceded by a comma unless it is the first one. The members of a member
+// annotated with flatten are written in its place.
+template <class W, class T>
+simdjson_really_inline void atom_fields(W &w, const T &t, bool &first) {
+  // Per-field block: ensure key+value worst case, then write key + value
+  // through the writer's local pos. For arithmetic fields, the integer
+  // write happens directly via write_uint_jeaiii on w.ptr+w.pos, so pos
+  // never round-trips through memory.
+  template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+    if constexpr (is_serialized_member(dm)) {
+      if (should_serialize<dm>(t.[:dm:])) {
+        if constexpr (simdjson::detail::has_annotation(dm, ^^simdjson::detail::flatten_tag)) {
+          static_assert(std::meta::is_class_type(simdjson::detail::flattened_type(dm)));
+          using flattened = std::remove_cvref_t<decltype(t.[:dm:])>;
+          static_assert(!concepts::container_but_not_string<flattened> && !concepts::string_view_keyed_map<flattened> &&
+                        !concepts::appendable_containers<flattened> && !concepts::optional_type<flattened> &&
+                        !concepts::smart_pointer<flattened> && !std::is_same_v<flattened, std::string> &&
+                        !std::is_same_v<flattened, std::string_view> && !require_custom_serialization<flattened>,
+                        "simdjson::flatten requires a member whose type is a structure serialized member by member");
+          atom_fields(w, t.[:dm:], first);
+        } else {
+          // Copy the key as whole 16-byte blocks from a zero-padded copy (one
+          // load and one store); ensure() reserves the padded length, and the
+          // unchecked writer has slack past its bound. Prior related work:
+          // jsonifier copies a power-of-two padded key and advances the cursor
+          // by the real length (serialize_impl.hpp, packed_blitter,
+          // https://github.com/nihilai-collective/Jsonifier).
+          constexpr const char* key_name = simdjson::get_json_key_name<dm>();
+          constexpr size_t first_key_len = constevalutil::consteval_to_quoted_escaped(key_name).size() + 1;
+          constexpr size_t rest_key_len = first_key_len + 1;
+          constexpr auto first_key = std::define_static_string(
+              constevalutil::consteval_to_quoted_escaped(key_name) + ":" +
+              std::string(padded_key_length(first_key_len) - first_key_len, '\0'));
+          constexpr auto rest_key = std::define_static_string(
+              std::string(",") + constevalutil::consteval_to_quoted_escaped(key_name) + ":" +
+              std::string(padded_key_length(rest_key_len) - rest_key_len, '\0'));
+          if (!w.ensure(padded_key_length(rest_key_len))) { return; }
+          if (first) {
+            std::memcpy(w.ptr + w.pos, first_key, padded_key_length(first_key_len));
+            w.pos += first_key_len;
+          } else {
+            std::memcpy(w.ptr + w.pos, rest_key, padded_key_length(rest_key_len));
+            w.pos += rest_key_len;
+          }
+          first = false;
+          atom_member<dm>(w, t.[:dm:]);
+        }
+      }
+    }
+  };
+}
+
+} // namespace annotation_detail
+
+template <class W, class T>
   requires(concepts::container_but_not_string<T> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &t) {
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
   auto it = t.begin();
   auto end = t.end();
   if (it == end) {
-    b.append_raw("[]");
+    if (!w.ensure(2)) return;
+    std::memcpy(w.ptr + w.pos, "[]", 2);
+    w.pos += 2;
     return;
   }
-  b.append('[');
-  atom(b, *it);
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = '[';
+  atom(w, *it);
   ++it;
   for (; it != end; ++it) {
-    b.append(',');
-    atom(b, *it);
+    if (!w.ensure(1)) return;
+    w.ptr[w.pos++] = ',';
+    atom(w, *it);
   }
-  b.append(']');
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = ']';
 }

-template <class T>
+template <class W, class T>
   requires(std::is_same_v<T, std::string> ||
            std::is_same_v<T, std::string_view> ||
            std::is_same_v<T, const char *> ||
            std::is_same_v<T, char>)
-constexpr void atom(string_builder &b, const T &t) {
-  b.escape_and_append_with_quotes(t);
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
+  // Inline the escape path through the writer so we never round-trip
+  // pos through memory for string fields (Twitter is dominated by
+  // these -- sync/reload around each string was a real cost).
+  std::string_view input;
+  if constexpr (std::is_same_v<T, char>) {
+    input = std::string_view(&t, 1);
+  } else {
+    input = std::string_view(t);
+  }
+  // Worst-case escape: every byte expands to \uXXXX (6 chars), plus 2 quotes.
+  // Guard against w.pos + 2 + 6 * input.size() wrapping for huge inputs -- if
+  // it wrapped to a small value, ensure() would spuriously succeed and the
+  // subsequent escape would overflow the buffer. max - w.pos cannot wrap, and
+  // size < (max - pos) / 6 implies pos + 6 * size + 6 <= max.
+  // Note that this is pedantic except maybe on 32-bit targets.
+  if constexpr (W::checked) {
+    if (simdjson_unlikely(input.size() >= ((std::numeric_limits<size_t>::max)() - w.pos) / 6)) { return; }
+    if (!w.ensure(2 + 6 * input.size())) { return; }
+  }
+  w.ptr[w.pos++] = '"';
+  w.pos += write_string_escaped(input, w.ptr + w.pos);
+  w.ptr[w.pos++] = '"';
 }

-template <concepts::string_view_keyed_map T>
+template <class W, concepts::string_view_keyed_map T>
   requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &m) {
+simdjson_really_inline constexpr void atom(W &w, const T &m) {
   if (m.empty()) {
-    b.append_raw("{}");
+    if (!w.ensure(2)) return;
+    std::memcpy(w.ptr + w.pos, "{}", 2);
+    w.pos += 2;
     return;
   }
-  b.append('{');
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = '{';
   bool first = true;
   for (const auto& [key, value] : m) {
     if (!first) {
-      b.append(',');
+      if (!w.ensure(1)) return;
+      w.ptr[w.pos++] = ',';
     }
     first = false;
-    // Keys must be convertible to string_view per the concept
-    b.escape_and_append_with_quotes(key);
-    b.append(':');
-    atom(b, value);
+    // Keys must be convertible to string_view per the concept.
+    std::string_view key_sv(key);
+    // Guard against w.pos + 3 + 6 * key_sv.size() wrapping for huge keys, if
+    // it wrapped to a small value, ensure() would spuriously succeed and the
+    // subsequent escape would overflow the buffer. max - w.pos cannot wrap, and
+    // size < (max - pos) / 6 implies pos + 6 * size + 6 <= max.
+    // Note that this is pedantic except maybe on 32-bit targets.
+    if constexpr (W::checked) {
+      if (simdjson_unlikely(key_sv.size() >= ((std::numeric_limits<size_t>::max)() - w.pos) / 6)) { return; }
+      if (!w.ensure(2 + 6 * key_sv.size() + 1)) { return; }
+    }
+    w.ptr[w.pos++] = '"';
+    w.pos += write_string_escaped(key_sv, w.ptr + w.pos);
+    w.ptr[w.pos++] = '"';
+    w.ptr[w.pos++] = ':';
+    atom(w, value);
   }
-  b.append('}');
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = '}';
 }


-template<typename number_type,
+template<class W, typename number_type,
          typename = typename std::enable_if<std::is_arithmetic<number_type>::value && !std::is_same_v<number_type, char>>::type>
-constexpr void atom(string_builder &b, const number_type t) {
-  b.append(t);
+simdjson_really_inline constexpr void atom(W &w, const number_type t) {
+  // Booleans / floats: defer to string_builder (rare path; keeps writer hot
+  // path free of float-formatter machinery). For integers, write directly
+  // via jeaiii using local pos.
+  if constexpr (std::is_same_v<number_type, bool>) {
+    if (t) {
+      if (!w.ensure(4)) return;
+      std::memcpy(w.ptr + w.pos, "true", 4);
+      w.pos += 4;
+    } else {
+      if (!w.ensure(5)) return;
+      std::memcpy(w.ptr + w.pos, "false", 5);
+      w.pos += 5;
+    }
+  } else if constexpr (std::is_floating_point_v<number_type>) {
+    if constexpr (W::checked) {
+      call_through_string_builder(w, [&](string_builder &b) { b.append(t); });
+    } else {
+      w.pos = size_t(internal::write_double(w.ptr + w.pos, double(t)) - w.ptr);
+    }
+  } else if constexpr (std::is_unsigned_v<number_type>) {
+    if (!w.ensure(20)) return;
+    char *end = internal::write_uint_jeaiii(
+        w.ptr + w.pos, static_cast<uint64_t>(t));
+    w.pos = static_cast<size_t>(end - w.ptr);
+  } else {
+    // signed integral
+    if (!w.ensure(20)) return;
+    using U = typename std::make_unsigned<number_type>::type;
+    bool negative = t < 0;
+    U pv = negative ? U(0) - static_cast<U>(t) : static_cast<U>(t);
+    w.ptr[w.pos] = '-';
+    w.pos += negative;
+    char *end = internal::write_uint_jeaiii(
+        w.ptr + w.pos, static_cast<uint64_t>(pv));
+    w.pos = static_cast<size_t>(end - w.ptr);
+  }
 }

-template <class T>
+template <class W, class T>
   requires(std::is_class_v<T> && !concepts::container_but_not_string<T> &&
            !concepts::string_view_keyed_map<T> &&
            !concepts::optional_type<T> &&
@@ -52429,92 +63767,259 @@ template <class T>
            !std::is_same_v<T, std::string_view> &&
            !std::is_same_v<T, const char*> &&
            !std::is_same_v<T, char> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &t) {
-  int i = 0;
-  b.append('{');
-  template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
-    if (i != 0)
-      b.append(',');
-    constexpr auto key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(dm)));
-    b.append_raw(key);
-    b.append(':');
-    atom(b, t.[:dm:]);
-    i++;
-  };
-  b.append('}');
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
+  if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag)) {
+    // A transparent structure is serialized as its single member.
+    constexpr auto dm = simdjson::detail::transparent_member(^^T);
+    annotation_detail::atom_member<dm>(w, t.[:dm:]);
+  } else {
+    bool first = true;
+    if (!w.ensure(1)) { return; }
+    w.ptr[w.pos++] = '{';
+    annotation_detail::atom_fields(w, t, first);
+    if (!w.ensure(1)) { return; }
+    w.ptr[w.pos++] = '}';
+  }
 }

 // Support for optional types (std::optional, etc.)
-template <concepts::optional_type T>
+template <class W, concepts::optional_type T>
   requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &opt) {
+simdjson_really_inline constexpr void atom(W &w, const T &opt) {
   if (opt) {
-    atom(b, opt.value());
+    atom(w, opt.value());
   } else {
-    b.append_raw("null");
+    if (!w.ensure(4)) return;
+    std::memcpy(w.ptr + w.pos, "null", 4);
+    w.pos += 4;
   }
 }

 // Support for smart pointers (std::unique_ptr, std::shared_ptr, etc.)
-template <concepts::smart_pointer T>
+template <class W, concepts::smart_pointer T>
   requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &ptr) {
+simdjson_really_inline constexpr void atom(W &w, const T &ptr) {
   if (ptr) {
-    atom(b, *ptr);
+    atom(w, *ptr);
   } else {
-    b.append_raw("null");
+    if (!w.ensure(4)) return;
+    std::memcpy(w.ptr + w.pos, "null", 4);
+    w.pos += 4;
   }
 }

 // Support for enums - serialize as string representation using expand approach from P2996R12
-template <typename T>
+template <class W, typename T>
   requires(std::is_enum_v<T> && !require_custom_serialization<T>)
-void atom(string_builder &b, const T &e) {
+simdjson_really_inline void atom(W &w, const T &e) {
 #if SIMDJSON_STATIC_REFLECTION
-  constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
+  static constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
   template for (constexpr auto enum_val : enumerators) {
-    constexpr auto enum_str = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(enum_val)));
+    constexpr auto enum_str = std::define_static_string(constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<enum_val>()));
+    constexpr size_t enum_str_len = std::char_traits<char>::length(enum_str);
     if (e == [:enum_val:]) {
-      b.append_raw(enum_str);
+      if (!w.ensure(enum_str_len)) return;
+      std::memcpy(w.ptr + w.pos, enum_str, enum_str_len);
+      w.pos += enum_str_len;
       return;
     }
   };
   // Fallback to integer if enum value not found
-  atom(b, static_cast<std::underlying_type_t<T>>(e));
+  atom(w, static_cast<std::underlying_type_t<T>>(e));
 #else
   // Fallback: serialize as integer if reflection not available
-  atom(b, static_cast<std::underlying_type_t<T>>(e));
+  atom(w, static_cast<std::underlying_type_t<T>>(e));
 #endif
 }

 // Support for appendable containers that don't have operator[] (sets, etc.)
-template <concepts::appendable_containers T>
+template <class W, concepts::appendable_containers T>
   requires(!concepts::container_but_not_string<T> && !concepts::string_view_keyed_map<T> &&
            !concepts::optional_type<T> && !concepts::smart_pointer<T> &&
            !std::is_same_v<T, std::string> &&
            !std::is_same_v<T, std::string_view> && !std::is_same_v<T, const char*> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &container) {
+simdjson_really_inline constexpr void atom(W &w, const T &container) {
   if (container.empty()) {
-    b.append_raw("[]");
+    if (!w.ensure(2)) return;
+    std::memcpy(w.ptr + w.pos, "[]", 2);
+    w.pos += 2;
     return;
   }
-  b.append('[');
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = '[';
   bool first = true;
   for (const auto& item : container) {
     if (!first) {
-      b.append(',');
+      if (!w.ensure(1)) return;
+      w.ptr[w.pos++] = ',';
     }
     first = false;
-    atom(b, item);
+    atom(w, item);
+  }
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = ']';
+}
+
+// =============================================================
+// Size bound: an upper bound on the number of bytes that atom(w, t) writes.
+// Computing it first lets append() reserve the capacity once and then run
+// the whole write chain through an unchecked_writer, without a capacity
+// check before every write. It mirrors the atom() overloads above.
+//
+// Prior related work: jsonifier sizes the document first, counting 6 bytes
+// per string byte, resizes once to that bound plus slack, and writes with
+// no capacity check on each store. to_json() does that resize through
+// std::string::resize_and_overwrite. See serializer.hpp and
+// serialize_impl.hpp in https://github.com/nihilai-collective/Jsonifier
+// and https://nihilai-collective.net/serialization.
+// =============================================================
+namespace bound_detail {
+
+// Whether size_bound covers everything that atom() writes for T: not when a
+// member is serialized by a with<Adapter> serializer, which writes an unknown
+// amount through the string_builder.
+template <class T>
+consteval bool is_bounded() {
+  if constexpr (require_custom_serialization<T>) {
+    return false;
+  } else if constexpr (std::is_same_v<T, std::string> || std::is_same_v<T, std::string_view> ||
+                       std::is_same_v<T, const char *> || std::is_arithmetic_v<T> || std::is_enum_v<T>) {
+    return true;
+  } else if constexpr (concepts::optional_type<T> || concepts::smart_pointer<T>) {
+    return is_bounded<std::remove_cvref_t<decltype(*std::declval<const T &>())>>();
+  } else if constexpr (concepts::string_view_keyed_map<T>) {
+    return is_bounded<std::remove_cvref_t<typename T::mapped_type>>();
+  } else if constexpr (concepts::container_but_not_string<T> || concepts::appendable_containers<T>) {
+    return is_bounded<std::remove_cvref_t<std::ranges::range_value_t<T>>>();
+  } else {
+    bool bounded = true;
+    template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+      if constexpr (annotation_detail::is_serialized_member(dm)) {
+        bounded = bounded && simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::with_t) == std::meta::info{} &&
+                  is_bounded<std::remove_cvref_t<decltype(std::declval<const T &>().[:dm:])>>();
+      }
+    };
+    return bounded;
+  }
+}
+
+template <class T>
+consteval size_t enum_bound() {
+  size_t bound = 20; // the integer fallback
+  template for (constexpr auto enum_val : std::define_static_array(std::meta::enumerators_of(^^T))) {
+    constexpr size_t len = std::char_traits<char>::length(std::define_static_string(
+        constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<enum_val>())));
+    bound = (std::max)(bound, len);
+  };
+  return bound;
+}
+
+template <class T>
+simdjson_really_inline size_t size_bound(const T &t) noexcept;
+
+// Bound for the "key":value pairs of a structure, commas included.
+template <class T>
+simdjson_really_inline size_t fields_bound(const T &t) noexcept {
+  size_t bound = 0;
+  template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+    if constexpr (annotation_detail::is_serialized_member(dm)) {
+      if constexpr (simdjson::detail::has_annotation(dm, ^^simdjson::detail::flatten_tag)) {
+        bound += fields_bound(t.[:dm:]);
+      } else {
+        constexpr size_t rest_key_len = constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<dm>()).size() + 2;
+        bound += rest_key_len + size_bound(t.[:dm:]);
+      }
+    }
+  };
+  return bound;
+}
+
+template <class T>
+simdjson_really_inline size_t size_bound([[maybe_unused]] const T &t) noexcept {
+  if constexpr (std::is_same_v<T, char>) {
+    return 2 + 6;
+  } else if constexpr (std::is_same_v<T, std::string> || std::is_same_v<T, std::string_view> ||
+                       std::is_same_v<T, const char *>) {
+    // Every byte may become \uXXXX, plus the quotes.
+    return 2 + 6 * std::string_view(t).size();
+  } else if constexpr (std::is_same_v<T, bool>) {
+    return 5;
+  } else if constexpr (std::is_floating_point_v<T>) {
+    return simdjson::internal::to_chars_buffer_size;
+  } else if constexpr (std::is_arithmetic_v<T>) {
+    return 20;
+  } else if constexpr (std::is_enum_v<T>) {
+    return enum_bound<T>();
+  } else if constexpr (concepts::optional_type<T> || concepts::smart_pointer<T>) {
+    return t ? size_bound(*t) : 4;
+  } else if constexpr (concepts::string_view_keyed_map<T>) {
+    size_t bound = 2;
+    for (const auto &[key, value] : t) {
+      // comma, quotes, colon
+      bound += 4 + 6 * std::string_view(key).size() + size_bound(value);
+    }
+    return bound;
+  } else if constexpr (concepts::container_but_not_string<T> || concepts::appendable_containers<T>) {
+    using value_type = std::remove_cvref_t<std::ranges::range_value_t<T>>;
+    if constexpr (std::is_arithmetic_v<value_type> && !std::is_same_v<value_type, char>) {
+      // A fixed bound per element: no need to visit them.
+      return 2 + size_t(std::ranges::distance(t)) * (1 + size_bound(value_type{}));
+    } else {
+      size_t bound = 2;
+#if defined(__GNUC__) && !defined(__clang__)
+#pragma GCC novector // a vector loop is slower on short containers
+#endif
+#pragma GCC unroll 4
+      for (const auto &item : t) {
+        bound += 1 + size_bound(item);
+      }
+      return bound;
+    }
+  } else if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag)) {
+    constexpr auto dm = simdjson::detail::transparent_member(^^T);
+    return size_bound(t.[:dm:]);
+  } else {
+    return 2 + fields_bound(t);
+  }
+}
+
+} // namespace bound_detail
+
+// Write t through an unchecked writer when its size bound is available,
+// reserving that many bytes first, and through the checked writer otherwise.
+template <class T>
+simdjson_really_inline void append_bounded(string_builder &b, const T &t) {
+  // On 32-bit systems, the bound could overflow: keep the checked writer.
+  if constexpr (sizeof(size_t) >= 8 && bound_detail::is_bounded<T>()) {
+    const size_t bound = bound_detail::size_bound(t) + unchecked_slack;
+    const size_t pos = b.unsafe_position();
+    // The bound is a sum of in-memory sizes times a small constant: it cannot
+    // overflow on a 64-bit system. Be pedantic elsewhere.
+    if (sizeof(size_t) >= 8 || bound <= (std::numeric_limits<size_t>::max)() - pos) {
+      const size_t cap = b.unsafe_capacity();
+      // Grow geometrically so that many small appends stay amortized.
+      if (pos + bound <= cap || b.unsafe_grow((std::max)(cap * 2, pos + bound))) {
+        unchecked_writer w(b.unsafe_data(), pos);
+        atom(w, t);
+        b.unsafe_set_position(w.pos);
+      }
+      return;
+    }
   }
-  b.append(']');
+  writer w(b);
+  atom(w, t);
+  w.sync();
 }

-// append functions that delegate to atom functions for primitive types
+// append() -- top-level entry. Each overload constructs a stack-local
+// writer, runs atom(w, t) through the inlined call chain, then syncs
+// the local position back into the string_builder.
 template <class T>
   requires(std::is_arithmetic_v<T> && !std::is_same_v<T, char>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  writer w(b);
+  atom(w, t);
+  w.sync();
 }

 template <class T>
@@ -52522,20 +64027,22 @@ template <class T>
            std::is_same_v<T, std::string_view> ||
            std::is_same_v<T, const char *> ||
            std::is_same_v<T, char>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  writer w(b);
+  atom(w, t);
+  w.sync();
 }

 template <concepts::optional_type T>
   requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  append_bounded(b, t);
 }

 template <concepts::smart_pointer T>
   requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  append_bounded(b, t);
 }

 template <concepts::appendable_containers T>
@@ -52543,14 +64050,14 @@ template <concepts::appendable_containers T>
            !concepts::optional_type<T> && !concepts::smart_pointer<T> &&
            !std::is_same_v<T, std::string> &&
            !std::is_same_v<T, std::string_view> && !std::is_same_v<T, const char*> && !require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  append_bounded(b, t);
 }

 template <concepts::string_view_keyed_map T>
   requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  append_bounded(b, t);
 }

 // works for struct
@@ -52564,39 +64071,15 @@ template <class Z>
            !std::is_same_v<Z, std::string_view> &&
            !std::is_same_v<Z, const char*> &&
            !std::is_same_v<Z, char> && !require_custom_serialization<Z>)
-void append(string_builder &b, const Z &z) {
-  int i = 0;
-  b.append('{');
-  template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^Z, std::meta::access_context::unchecked()))) {
-    if (i != 0)
-      b.append(',');
-    constexpr auto key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(dm)));
-    b.append_raw(key);
-    b.append(':');
-    atom(b, z.[:dm:]);
-    i++;
-  };
-  b.append('}');
+simdjson_inline void append(string_builder &b, const Z &z) {
+  append_bounded(b, z);
 }

 // works for container that have begin() and end() iterators
 template <class Z>
   requires(concepts::container_but_not_string<Z> && !require_custom_serialization<Z>)
-void append(string_builder &b, const Z &z) {
-  auto it = z.begin();
-  auto end = z.end();
-  if (it == end) {
-    b.append_raw("[]");
-    return;
-  }
-  b.append('[');
-  atom(b, *it);
-  ++it;
-  for (; it != end; ++it) {
-    b.append(',');
-    atom(b, *it);
-  }
-  b.append(']');
+simdjson_inline void append(string_builder &b, const Z &z) {
+  append_bounded(b, z);
 }

 template <class Z>
@@ -52607,22 +64090,40 @@ void append(string_builder &b, const Z &z) {


 template <class Z>
-simdjson_warn_unused simdjson_result<std::string> to_json_string(const Z &z, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
-  string_builder b(initial_capacity);
-  append(b, z);
-  std::string_view s;
-  if(auto e = b.view().get(s); e) { return e; }
-  return std::string(s);
+simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
+  if constexpr (sizeof(size_t) >= 8 && bound_detail::is_bounded<Z>()) {
+    // Write straight into s, sized by the bound: no intermediate buffer, no copy.
+    // Prior related work: jsonifier's serializeJson resizes once through
+    // resize_and_overwrite (serializer.hpp).
+    (void)initial_capacity;
+    const size_t bound = bound_detail::size_bound(z) + unchecked_slack;
+    auto write = [&z](char *p) noexcept {
+      unchecked_writer w(p, 0);
+      atom(w, z);
+      return w.pos;
+    };
+#if defined(__cpp_lib_string_resize_and_overwrite) && __cpp_lib_string_resize_and_overwrite >= 202110L
+    s.resize_and_overwrite(bound, [&write](char *p, size_t) noexcept { return write(p); });
+#else
+    s.resize(bound);
+    s.resize(write(s.data()));
+#endif
+    return SUCCESS;
+  } else {
+    string_builder b(initial_capacity);
+    append(b, z);
+    std::string_view view;
+    if(auto e = b.view().get(view); e) { return e; }
+    s.assign(view);
+    return SUCCESS;
+  }
 }

 template <class Z>
-simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
-  string_builder b(initial_capacity);
-  append(b, z);
-  std::string_view view;
-  if(auto e = b.view().get(view); e) { return e; }
-  s.assign(view);
-  return SUCCESS;
+simdjson_warn_unused simdjson_result<std::string> to_json_string(const Z &z, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
+  std::string s;
+  if(auto e = to_json(z, s, initial_capacity); e) { return e; }
+  return s;
 }

 template <class Z>
@@ -52635,40 +64136,41 @@ string_builder& operator<<(string_builder& b, const Z& z) {
 template<constevalutil::fixed_string... FieldNames, typename T>
   requires(std::is_class_v<T> && (sizeof...(FieldNames) > 0))
 void extract_from(string_builder &b, const T &obj) {
-  // Helper to check if a field name matches any of the requested fields
-  auto should_extract = [](std::string_view field_name) constexpr -> bool {
-    return ((FieldNames.view() == field_name) || ...);
-  };
-
-  b.append('{');
+  writer w(b);
+  if (!w.ensure(1)) { w.sync(); return; }
+  w.ptr[w.pos++] = '{';
   bool first = true;
-
   // Iterate through all members of T using reflection
-  template for (constexpr auto mem : std::define_static_array(
-      std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
-
+  static constexpr auto members = std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()));
+  template for (constexpr auto mem : members) {
     if constexpr (std::meta::is_public(mem)) {
-      constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
+      static constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));

       // Only serialize this field if it's in our list of requested fields
-      if constexpr (should_extract(key)) {
-        if (!first) {
-          b.append(',');
+      if constexpr (((FieldNames.view() == key) || ...)) {
+        static constexpr auto first_key = std::define_static_string(
+            constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)) + ":");
+        static constexpr auto rest_key = std::define_static_string(
+            std::string(",") + constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)) + ":");
+        constexpr size_t first_key_len = std::char_traits<char>::length(first_key);
+        constexpr size_t rest_key_len = std::char_traits<char>::length(rest_key);
+        if (!w.ensure(rest_key_len)) { w.sync(); return; }
+        if (first) {
+          std::memcpy(w.ptr + w.pos, first_key, first_key_len);
+          w.pos += first_key_len;
+        } else {
+          std::memcpy(w.ptr + w.pos, rest_key, rest_key_len);
+          w.pos += rest_key_len;
         }
         first = false;
-
-        // Serialize the key
-        constexpr auto quoted_key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)));
-        b.append_raw(quoted_key);
-        b.append(':');
-
-        // Serialize the value
-        atom(b, obj.[:mem:]);
+        atom(w, obj.[:mem:]);
       }
     }
   };

-  b.append('}');
+  if (!w.ensure(1)) { w.sync(); return; }
+  w.ptr[w.pos++] = '}';
+  w.sync();
 }

 template<constevalutil::fixed_string... FieldNames, typename T>
@@ -52681,25 +64183,18 @@ simdjson_warn_unused simdjson_result<std::string> extract_from(const T &obj, siz
   return std::string(s);
 }

+SIMDJSON_POP_DISABLE_WARNINGS
+
 } // namespace builder
 } // namespace westmere
 // Alias the function template to 'to' in the global namespace
 template <class Z>
 simdjson_warn_unused simdjson_result<std::string> to_json(const Z &z, size_t initial_capacity = westmere::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
-  westmere::builder::string_builder b(initial_capacity);
-  westmere::builder::append(b, z);
-  std::string_view s;
-  if(auto e = b.view().get(s); e) { return e; }
-  return std::string(s);
+  return westmere::builder::to_json_string(z, initial_capacity);
 }
 template <class Z>
 simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = westmere::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
-  westmere::builder::string_builder b(initial_capacity);
-  westmere::builder::append(b, z);
-  std::string_view view;
-  if(auto e = b.view().get(view); e) { return e; }
-  s.assign(view);
-  return SUCCESS;
+  return westmere::builder::to_json(z, s, initial_capacity);
 }
 // Global namespace function for extract_from
 template<constevalutil::fixed_string... FieldNames, typename T>
@@ -52845,6 +64340,7 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
 /* including simdjson/generic/builder/json_string_builder-inl.h for westmere: #include "simdjson/generic/builder/json_string_builder-inl.h" */
 /* begin file simdjson/generic/builder/json_string_builder-inl.h for westmere */
 #include <array>
+#include <cmath>
 #include <cstring>
 #include <limits>
 #include <type_traits>
@@ -52879,6 +64375,11 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
 #define SIMDJSON_EXPERIMENTAL_HAS_LSX 1
 #endif
 #endif
+#if defined(__loongarch_asx)
+#ifndef SIMDJSON_EXPERIMENTAL_HAS_LASX
+#define SIMDJSON_EXPERIMENTAL_HAS_LASX 1
+#endif
+#endif
 #if defined(__riscv_v_intrinsic) && __riscv_v_intrinsic >= 11000 &&            \
     defined(__riscv_vector)
 #ifndef SIMDJSON_EXPERIMENTAL_HAS_RVV
@@ -52898,6 +64399,9 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
 #endif
 #if SIMDJSON_EXPERIMENTAL_HAS_SSE2
 #include <emmintrin.h>
+#if defined(__AVX2__)
+#include <immintrin.h>
+#endif
 #ifdef _MSC_VER
 #include <intrin.h>
 #endif
@@ -52905,6 +64409,9 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
 #if SIMDJSON_EXPERIMENTAL_HAS_LSX
 #include <lsxintrin.h>
 #endif
+#if SIMDJSON_EXPERIMENTAL_HAS_LASX
+#include <lasxintrin.h>
+#endif
 #if SIMDJSON_EXPERIMENTAL_HAS_RVV
 #include <riscv_vector.h>
 #endif
@@ -52954,105 +64461,6 @@ inline bool has_json_escapable_byte(uint64_t x) {

 **/

-SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline bool
-simple_needs_escaping(std::string_view v) {
-  for (char c : v) {
-    // a table lookup is faster than a series of comparisons
-    if (json_quotable_character[static_cast<uint8_t>(c)]) {
-      return true;
-    }
-  }
-  return false;
-}
-
-#if SIMDJSON_EXPERIMENTAL_HAS_NEON
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
-  if (view.size() < 16) {
-    return simple_needs_escaping(view);
-  }
-  size_t i = 0;
-  uint8x16_t running = vdupq_n_u8(0);
-  uint8x16_t v34 = vdupq_n_u8(34);
-  uint8x16_t v92 = vdupq_n_u8(92);
-
-  for (; i + 15 < view.size(); i += 16) {
-    uint8x16_t word = vld1q_u8((const uint8_t *)view.data() + i);
-    running = vorrq_u8(running, vceqq_u8(word, v34));
-    running = vorrq_u8(running, vceqq_u8(word, v92));
-    running = vorrq_u8(running, vcltq_u8(word, vdupq_n_u8(32)));
-  }
-  if (i < view.size()) {
-    uint8x16_t word =
-        vld1q_u8((const uint8_t *)view.data() + view.length() - 16);
-    running = vorrq_u8(running, vceqq_u8(word, v34));
-    running = vorrq_u8(running, vceqq_u8(word, v92));
-    running = vorrq_u8(running, vcltq_u8(word, vdupq_n_u8(32)));
-  }
-  return vmaxvq_u32(vreinterpretq_u32_u8(running)) != 0;
-}
-#elif SIMDJSON_EXPERIMENTAL_HAS_SSE2
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
-  if (view.size() < 16) {
-    return simple_needs_escaping(view);
-  }
-  size_t i = 0;
-  __m128i running = _mm_setzero_si128();
-  for (; i + 15 < view.size(); i += 16) {
-
-    __m128i word =
-        _mm_loadu_si128(reinterpret_cast<const __m128i *>(view.data() + i));
-    running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(34)));
-    running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(92)));
-    running = _mm_or_si128(
-        running, _mm_cmpeq_epi8(_mm_subs_epu8(word, _mm_set1_epi8(31)),
-                                _mm_setzero_si128()));
-  }
-  if (i < view.size()) {
-    __m128i word = _mm_loadu_si128(
-        reinterpret_cast<const __m128i *>(view.data() + view.length() - 16));
-    running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(34)));
-    running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(92)));
-    running = _mm_or_si128(
-        running, _mm_cmpeq_epi8(_mm_subs_epu8(word, _mm_set1_epi8(31)),
-                                _mm_setzero_si128()));
-  }
-  return _mm_movemask_epi8(running) != 0;
-}
-#elif SIMDJSON_EXPERIMENTAL_HAS_PPC64
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
-  if (view.size() < 16) {
-    return simple_needs_escaping(view);
-  }
-  size_t i = 0;
-  __vector unsigned char running = vec_splats((unsigned char)0);
-  __vector unsigned char v34 = vec_splats((unsigned char)34);
-  __vector unsigned char v92 = vec_splats((unsigned char)92);
-  __vector unsigned char v32 = vec_splats((unsigned char)32);
-
-  for (; i + 15 < view.size(); i += 16) {
-    __vector unsigned char word =
-        vec_vsx_ld(0, reinterpret_cast<const unsigned char *>(view.data() + i));
-    running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v34));
-    running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v92));
-    running = vec_or(running,
-        (__vector unsigned char)vec_cmplt(word, v32));
-  }
-  if (i < view.size()) {
-    __vector unsigned char word = vec_vsx_ld(
-        0, reinterpret_cast<const unsigned char *>(view.data() + view.length() - 16));
-    running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v34));
-    running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v92));
-    running = vec_or(running,
-        (__vector unsigned char)vec_cmplt(word, v32));
-  }
-  return !vec_all_eq(running, vec_splats((unsigned char)0));
-}
-#else
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
-  return simple_needs_escaping(view);
-}
-#endif
-
 // Scalar fallback for finding next quotable character
 SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline size_t
 find_next_json_quotable_character_scalar(const std::string_view view,
@@ -53143,6 +64551,51 @@ find_next_json_quotable_character(const std::string_view view,
   size_t current = len - remaining;
   return find_next_json_quotable_character_scalar(view, current);
 }
+#elif SIMDJSON_EXPERIMENTAL_HAS_LASX
+simdjson_inline size_t
+find_next_json_quotable_character(const std::string_view view,
+                                  size_t location) noexcept {
+  const size_t len = view.size();
+  const uint8_t *ptr =
+      reinterpret_cast<const uint8_t *>(view.data()) + location;
+  size_t remaining = len - location;
+
+  // SIMD constants for characters requiring escape
+  __m256i v34 = __lasx_xvreplgr2vr_b(34);  // '"'
+  __m256i v92 = __lasx_xvreplgr2vr_b(92);  // '\\'
+  __m256i v32 = __lasx_xvreplgr2vr_b(32);  // control char threshold
+
+  while (remaining >= 32) {
+    __m256i word = __lasx_xvld(ptr, 0);
+
+    // Check for quotable characters: '"', '\\', or control chars (< 32)
+    __m256i needs_escape = __lasx_xvseq_b(word, v34);
+    needs_escape = __lasx_xvor_v(needs_escape, __lasx_xvseq_b(word, v92));
+    needs_escape = __lasx_xvor_v(needs_escape, __lasx_xvslt_bu(word, v32));
+
+    if (!__lasx_xbz_v(needs_escape)) {
+      // Found a quotable character - locate it via the four 64-bit lanes
+      uint64_t lane0 = __lasx_xvpickve2gr_du(needs_escape, 0);
+      uint64_t lane1 = __lasx_xvpickve2gr_du(needs_escape, 1);
+      uint64_t lane2 = __lasx_xvpickve2gr_du(needs_escape, 2);
+      uint64_t lane3 = __lasx_xvpickve2gr_du(needs_escape, 3);
+      size_t offset = ptr - reinterpret_cast<const uint8_t *>(view.data());
+      if (lane0 != 0) {
+        return offset + trailing_zeroes(lane0) / 8;
+      } else if (lane1 != 0) {
+        return offset + 8 + trailing_zeroes(lane1) / 8;
+      } else if (lane2 != 0) {
+        return offset + 16 + trailing_zeroes(lane2) / 8;
+      } else {
+        return offset + 24 + trailing_zeroes(lane3) / 8;
+      }
+    }
+    ptr += 32;
+    remaining -= 32;
+  }
+  size_t current = len - remaining;
+  return find_next_json_quotable_character_scalar(view, current);
+}
 #elif SIMDJSON_EXPERIMENTAL_HAS_LSX
 simdjson_inline size_t
 find_next_json_quotable_character(const std::string_view view,
@@ -53298,6 +64751,254 @@ SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline void escape_json_char(char c, char *&o
   }
 }

+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2 || SIMDJSON_EXPERIMENTAL_HAS_NEON
+#define SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE 1
+#endif
+
+#if SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2
+
+using escape_vector = __m128i;
+// Mask bits that each input byte contributes to escape_bitmask().
+static constexpr unsigned escape_mask_bits = 1;
+
+simdjson_inline escape_vector escape_load16(const uint8_t *p) noexcept {
+  return _mm_loadu_si128(reinterpret_cast<const __m128i *>(p));
+}
+
+simdjson_inline void escape_store16(char *out, escape_vector v) noexcept {
+  _mm_storeu_si128(reinterpret_cast<__m128i *>(out), v);
+}
+
+// Builds a vector whose bytes 0..7 come from a and bytes 8..15 from b.
+simdjson_inline escape_vector escape_load8x2(const uint8_t *a,
+                                             const uint8_t *b) noexcept {
+  return _mm_unpacklo_epi64(
+      _mm_loadl_epi64(reinterpret_cast<const __m128i *>(a)),
+      _mm_loadl_epi64(reinterpret_cast<const __m128i *>(b)));
+}
+
+// Builds a vector whose bytes 0..3 come from a and bytes 4..7 from b. The
+// upper half is unspecified; callers only look at the low eight bits.
+simdjson_inline escape_vector escape_load4x2(const uint8_t *a,
+                                             const uint8_t *b) noexcept {
+  int32_t a32, b32;
+  memcpy(&a32, a, 4);
+  memcpy(&b32, b, 4);
+  return _mm_unpacklo_epi32(_mm_cvtsi32_si128(a32), _mm_cvtsi32_si128(b32));
+}
+
+// Sets every bit of byte k when byte k of v is a quotable character ('"',
+// '\\' or a control character).
+simdjson_inline escape_vector escape_flags(escape_vector v) noexcept {
+  const __m128i v34 = _mm_set1_epi8(34); // '"'
+  const __m128i v92 = _mm_set1_epi8(92); // '\\'
+  const __m128i v31 = _mm_set1_epi8(31); // for control char detection
+  __m128i needs_escape = _mm_cmpeq_epi8(v, v34);
+  needs_escape = _mm_or_si128(needs_escape, _mm_cmpeq_epi8(v, v92));
+  return _mm_or_si128(
+      needs_escape, _mm_cmpeq_epi8(_mm_subs_epu8(v, v31), _mm_setzero_si128()));
+}
+
+// True when any byte needs escaping. Kept separate from escape_bitmask
+// because some instruction sets can answer it without leaving the vector
+// register file; on SSE2 the compiler folds the two together.
+simdjson_inline bool escape_any(escape_vector flags) noexcept {
+  return _mm_movemask_epi8(flags) != 0;
+}
+
+// A 16-bit mask with bit k set when byte k needed escaping.
+simdjson_inline uint64_t escape_bitmask(escape_vector flags) noexcept {
+  return uint64_t(uint32_t(_mm_movemask_epi8(flags)));
+}
+
+#else // SIMDJSON_EXPERIMENTAL_HAS_NEON
+
+using escape_vector = uint8x16_t;
+static constexpr unsigned escape_mask_bits = 4;
+
+simdjson_inline escape_vector escape_load16(const uint8_t *p) noexcept {
+  return vld1q_u8(p);
+}
+
+simdjson_inline void escape_store16(char *out, escape_vector v) noexcept {
+  vst1q_u8(reinterpret_cast<uint8_t *>(out), v);
+}
+
+simdjson_inline escape_vector escape_load8x2(const uint8_t *a,
+                                             const uint8_t *b) noexcept {
+  uint64_t a64, b64;
+  memcpy(&a64, a, 8);
+  memcpy(&b64, b, 8);
+  return vreinterpretq_u8_u64(vsetq_lane_u64(b64, vdupq_n_u64(a64), 1));
+}
+
+simdjson_inline escape_vector escape_load4x2(const uint8_t *a,
+                                             const uint8_t *b) noexcept {
+  uint32_t a32, b32;
+  memcpy(&a32, a, 4);
+  memcpy(&b32, b, 4);
+  return vreinterpretq_u8_u32(vsetq_lane_u32(b32, vdupq_n_u32(a32), 1));
+}
+
+simdjson_inline escape_vector escape_flags(escape_vector v) noexcept {
+  uint8x16_t needs_escape = vceqq_u8(v, vdupq_n_u8(34));              // '"'
+  needs_escape = vorrq_u8(needs_escape, vceqq_u8(v, vdupq_n_u8(92))); // '\\'
+  return vorrq_u8(needs_escape, vcltq_u8(v, vdupq_n_u8(32)));
+}
+
+simdjson_inline bool escape_any(escape_vector flags) noexcept {
+  return vmaxvq_u32(vreinterpretq_u32_u8(flags)) != 0;
+}
+
+// Four bits per byte rather than one: escape_block and the tail paths scale
+// their shifts by escape_mask_bits to match.
+simdjson_inline uint64_t escape_bitmask(escape_vector flags) noexcept {
+  uint8x8_t narrowed = vshrn_n_u16(vreinterpretq_u16_u8(flags), 4);
+  return vget_lane_u64(vreinterpret_u64_u8(narrowed), 0);
+}
+
+#endif // instruction set selection
+
+// The escape bitmask of a 16-byte block.
+simdjson_inline uint64_t escape_mask(escape_vector v) noexcept {
+  return escape_bitmask(escape_flags(v));
+}
+
+// Copies n bytes with n < 16, using overlapping loads and stores. It never
+// reads more than n bytes from src, nor writes more than n bytes to dst.
+simdjson_inline void copy_lt16(char *dst, const uint8_t *src,
+                               size_t n) noexcept {
+  if (n >= 8) {
+    memcpy(dst, src, 8);
+    memcpy(dst + n - 8, src + n - 8, 8);
+  } else if (n >= 4) {
+    memcpy(dst, src, 4);
+    memcpy(dst + n - 4, src + n - 4, 4);
+  } else if (n > 0) {
+    dst[0] = char(src[0]);
+    dst[n >> 1] = char(src[n >> 1]);
+    dst[n - 1] = char(src[n - 1]);
+  }
+}
+
+// Escapes the bytes of src in the range [i, blockend), given that m is the
+// (non-zero) escape mask for that range: the escape_mask_bits-wide lane of m
+// at byte k is non-zero when src[i + k] requires escaping. Returns the updated
+// output pointer.
+simdjson_never_inline char *escape_block(const uint8_t *src, char *out,
+                                         size_t i, size_t blockend,
+                                         uint64_t m) noexcept {
+  constexpr uint64_t lane = (uint64_t(1) << escape_mask_bits) - 1;
+
+  size_t pos = i; // first byte not yet copied
+  while (m) {
+    const size_t tz = trailing_zeroes(m);
+    const size_t next = i + tz / escape_mask_bits;
+    // Copy the run of safe bytes that precedes this escape.
+    copy_lt16(out, src + pos, next - pos);
+    out += next - pos;
+    escape_json_char(char(src[next]), out);
+    pos = next + 1;
+    m &= ~(lane << tz);
+  }
+  // Copy whatever follows the last escape.
+  copy_lt16(out, src + pos, blockend - pos);
+  return out + (blockend - pos);
+}
+
+// Writes the escaped version of input to out, returning the number of bytes
+// written.
+simdjson_really_inline size_t write_string_escaped(const std::string_view input, char *out) {
+  const size_t len = input.size();
+  const uint8_t *src = reinterpret_cast<const uint8_t *>(input.data());
+  const char *const initout = out;
+
+  size_t i = 0;
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2 && defined(__AVX2__)
+  while (i + 32 <= len) {
+    const __m256i word = _mm256_loadu_si256(reinterpret_cast<const __m256i *>(src + i));
+    const __m256i flags = _mm256_or_si256(
+        _mm256_or_si256(_mm256_cmpeq_epi8(word, _mm256_set1_epi8(34)),   // '"'
+                        _mm256_cmpeq_epi8(word, _mm256_set1_epi8(92))),  // '\\'
+        _mm256_cmpeq_epi8(_mm256_subs_epu8(word, _mm256_set1_epi8(31)),
+                          _mm256_setzero_si256()));                      // control
+    const uint32_t mask = uint32_t(_mm256_movemask_epi8(flags));
+    if (simdjson_likely(mask == 0)) {
+      _mm256_storeu_si256(reinterpret_cast<__m256i *>(out), word);
+      out += 32;
+    } else {
+      for (size_t half = 0; half < 32; half += 16) {
+        const uint64_t m = (mask >> half) & 0xFFFF;
+        if (m == 0) {
+          escape_store16(out, escape_load16(src + i + half));
+          out += 16;
+        } else {
+          out = escape_block(src, out, i + half, i + half + 16, m);
+        }
+      }
+    }
+    i += 32;
+  }
+#endif
+  while (i + 16 <= len) {
+    escape_vector word = escape_load16(src + i);
+    escape_vector flags = escape_flags(word);
+    if (simdjson_likely(!escape_any(flags))) {
+      escape_store16(out, word);
+      out += 16;
+    } else {
+      out = escape_block(src, out, i, i + 16, escape_bitmask(flags));
+    }
+    i += 16;
+  }
+  if (i < len) {
+    const size_t rem = len - i;
+    uint64_t m;
+    if (len >= 16) {
+      // The last 16 bytes of the input are in bounds. Bit k of that block's
+      // mask belongs to input position len - 16 + k, so shift it down to align
+      // bit 0 with position i.
+      m = escape_mask(escape_load16(src + len - 16)) >>
+          (escape_mask_bits * (16 - rem));
+    } else if (len >= 8) {
+      // Here i == 0 and rem == len. Two overlapping 8-byte loads cover
+      // [0, 8) and [len - 8, len), which is the whole input since len < 16.
+      uint64_t mm = escape_mask(escape_load8x2(src, src + len - 8));
+      constexpr uint64_t low8 = (uint64_t(1) << (escape_mask_bits * 8)) - 1;
+      m = (mm & low8) |
+          ((mm >> (escape_mask_bits * 8)) << (escape_mask_bits * (len - 8)));
+    } else if (len >= 4) {
+      // Same idea with two overlapping 4-byte loads.
+      uint64_t mm = escape_mask(escape_load4x2(src, src + len - 4));
+      constexpr uint64_t low4 = (uint64_t(1) << (escape_mask_bits * 4)) - 1;
+      m = (mm & low4) | (((mm >> (escape_mask_bits * 4)) & low4)
+                         << (escape_mask_bits * (len - 4)));
+    } else {
+      // Fewer than 4 bytes: at most three table lookups, no need for SIMD.
+      for (size_t k = 0; k < len; k++) {
+        uint8_t c = src[k];
+        if (json_quotable_character[c]) {
+          escape_json_char(char(c), out);
+        } else {
+          *out++ = char(c);
+        }
+      }
+      return size_t(out - initout);
+    }
+    if (m == 0) {
+      copy_lt16(out, src + i, rem);
+      out += rem;
+    } else {
+      out = escape_block(src, out, i, len, m);
+    }
+  }
+  return size_t(out - initout);
+}
+
+#else // SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
 // Writes the escaped version of input to out, returning the number of bytes
 // written. Uses SIMD position finding to locate quotable characters efficiently.
 inline size_t write_string_escaped(const std::string_view input, char *out) {
@@ -53327,9 +65028,13 @@ inline size_t write_string_escaped(const std::string_view input, char *out) {
     escape_json_char(input[location], out);
     location += 1;
   }
-  return out - initout;
+  return size_t(out - initout);
 }

+#endif // SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+#undef SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
+
 simdjson_inline string_builder::string_builder(size_t initial_capacity)
     : buffer(new(std::nothrow) char[initial_capacity]), position(0),
       capacity(buffer.get() != nullptr ? initial_capacity : 0),
@@ -53352,7 +65057,7 @@ simdjson_inline bool string_builder::capacity_check(size_t upcoming_bytes) {
   return is_valid;
 }

-simdjson_inline void string_builder::grow_buffer(size_t desired_capacity) {
+inline void string_builder::grow_buffer(size_t desired_capacity) {
   if (!is_valid) {
     return;
   }
@@ -53408,81 +65113,136 @@ simdjson_inline void string_builder::clear() noexcept {

 namespace internal {

-template <typename number_type, typename = typename std::enable_if<
-                                    std::is_unsigned<number_type>::value>::type>
-simdjson_really_inline int int_log2(number_type x) {
-  return 63 - leading_zeroes(uint64_t(x) | 1);
-}
-
-simdjson_really_inline int fast_digit_count_32(uint32_t x) {
-  static uint64_t table[] = {
-      4294967296,  8589934582,  8589934582,  8589934582,  12884901788,
-      12884901788, 12884901788, 17179868184, 17179868184, 17179868184,
-      21474826480, 21474826480, 21474826480, 21474826480, 25769703776,
-      25769703776, 25769703776, 30063771072, 30063771072, 30063771072,
-      34349738368, 34349738368, 34349738368, 34349738368, 38554705664,
-      38554705664, 38554705664, 41949672960, 41949672960, 41949672960,
-      42949672960, 42949672960};
-  return uint32_t((x + table[int_log2(x)]) >> 32);
-}
-
-simdjson_really_inline int fast_digit_count_64(uint64_t x) {
-  static uint64_t table[] = {9,
-                             99,
-                             999,
-                             9999,
-                             99999,
-                             999999,
-                             9999999,
-                             99999999,
-                             999999999,
-                             9999999999,
-                             99999999999,
-                             999999999999,
-                             9999999999999,
-                             99999999999999,
-                             999999999999999ULL,
-                             9999999999999999ULL,
-                             99999999999999999ULL,
-                             999999999999999999ULL,
-                             9999999999999999999ULL};
-  int y = (19 * int_log2(x) >> 6);
-  y += x > table[y];
-  return y + 1;
-}
-
-template <typename number_type, typename = typename std::enable_if<
-                                    std::is_unsigned<number_type>::value>::type>
-simdjson_really_inline size_t digit_count(number_type v) noexcept {
-  static_assert(sizeof(number_type) == 8 || sizeof(number_type) == 4 ||
-                    sizeof(number_type) == 2 || sizeof(number_type) == 1,
-                "We only support 8-bit, 16-bit, 32-bit and 64-bit numbers");
-  SIMDJSON_IF_CONSTEXPR(sizeof(number_type) <= 4) {
-    return fast_digit_count_32(static_cast<uint32_t>(v));
+// Integer to decimal: James Edward Anhalt III's algorithm
+static const char jeaiii_dd[201] =
+    "00010203040506070809101112131415161718192021222324252627282930313233343536373839"
+    "40414243444546474849505152535455565758596061626364656667686970717273747576777879"
+    "8081828384858687888990919293949596979899";
+static const char jeaiii_fd[201] =
+    "0\0" "1\0" "2\0" "3\0" "4\0" "5\0" "6\0" "7\0" "8\0" "9\0"
+    "10111213141516171819202122232425262728293031323334353637383940414243444546474849"
+    "50515253545556575859606162636465666768697071727374757677787980818283848586878889"
+    "90919293949596979899";
+
+simdjson_really_inline void jeaiii_write_dd(char *p, uint64_t k) noexcept {
+  std::memcpy(p, &jeaiii_dd[2 * k], 2);
+}
+simdjson_really_inline void jeaiii_write_fd(char *p, uint64_t k) noexcept {
+  std::memcpy(p, &jeaiii_fd[2 * k], 2);
+}
+
+// Caller guarantees n < 10^8. Writes 1 to 8 digits.
+simdjson_really_inline char *jeaiii_lt1e8(char *b, uint32_t n) noexcept {
+  constexpr uint64_t mask24 = (uint64_t(1) << 24) - 1;
+  constexpr uint64_t mask32 = (uint64_t(1) << 32) - 1;
+  if (n < 100) {
+    jeaiii_write_fd(b, n);
+    return n < 10 ? b + 1 : b + 2;
+  }
+  if (n < 1000000) {
+    if (n < 10000) {
+      const uint32_t f0 = uint32_t(10 * (1 << 24) / 1e3 + 1) * n;
+      jeaiii_write_fd(b, f0 >> 24);
+      b -= n < 1000;
+      const uint32_t f2 = uint32_t(f0 & mask24) * 100;
+      jeaiii_write_dd(b + 2, f2 >> 24);
+      return b + 4;
+    }
+    const uint64_t f0 = uint64_t(10 * (1ull << 32) / 1e5 + 1) * n;
+    jeaiii_write_fd(b, f0 >> 32);
+    b -= n < 100000;
+    const uint64_t f2 = (f0 & mask32) * 100;
+    jeaiii_write_dd(b + 2, f2 >> 32);
+    const uint64_t f4 = (f2 & mask32) * 100;
+    jeaiii_write_dd(b + 4, f4 >> 32);
+    return b + 6;
+  }
+  const uint64_t f0 = uint64_t(10 * (1ull << 48) / 1e7 + 1) * n >> 16;
+  jeaiii_write_fd(b, f0 >> 32);
+  b -= n < 10000000;
+  const uint64_t f2 = (f0 & mask32) * 100;
+  jeaiii_write_dd(b + 2, f2 >> 32);
+  const uint64_t f4 = (f2 & mask32) * 100;
+  jeaiii_write_dd(b + 4, f4 >> 32);
+  const uint64_t f6 = (f4 & mask32) * 100;
+  jeaiii_write_dd(b + 6, f6 >> 32);
+  return b + 8;
+}
+
+// Caller guarantees z < 10^8. Always writes exactly 8 digits.
+simdjson_really_inline char *jeaiii_8_digits(char *b, uint32_t z) noexcept {
+  constexpr uint64_t mask32 = (uint64_t(1) << 32) - 1;
+  const uint64_t f0 = (uint64_t((1ull << 48) / 1e6 + 1) * z >> 16) + 1;
+  jeaiii_write_dd(b, f0 >> 32);
+  const uint64_t f2 = (f0 & mask32) * 100;
+  jeaiii_write_dd(b + 2, f2 >> 32);
+  const uint64_t f4 = (f2 & mask32) * 100;
+  jeaiii_write_dd(b + 4, f4 >> 32);
+  const uint64_t f6 = (f4 & mask32) * 100;
+  jeaiii_write_dd(b + 6, f6 >> 32);
+  return b + 8;
+}
+
+// Caller guarantees 10^8 <= n < 2^32. Writes 9 or 10 digits.
+simdjson_really_inline char *jeaiii_9_or_10(char *b, uint64_t n) noexcept {
+  constexpr uint64_t mask57 = (uint64_t(1) << 57) - 1;
+  const uint64_t f0 = uint64_t(10 * (1ull << 57) / 1e9 + 1) * n;
+  jeaiii_write_fd(b, f0 >> 57);
+  b -= n < 1000000000;
+  const uint64_t f2 = (f0 & mask57) * 100;
+  jeaiii_write_dd(b + 2, f2 >> 57);
+  const uint64_t f4 = (f2 & mask57) * 100;
+  jeaiii_write_dd(b + 4, f4 >> 57);
+  const uint64_t f6 = (f4 & mask57) * 100;
+  jeaiii_write_dd(b + 6, f6 >> 57);
+  const uint64_t f8 = (f6 & mask57) * 100;
+  jeaiii_write_dd(b + 8, f8 >> 57);
+  return b + 10;
+}
+
+simdjson_really_inline char *write_uint_jeaiii(char *b, uint64_t n) noexcept {
+  if (n < 100000000) {
+    return jeaiii_lt1e8(b, uint32_t(n));
+  }
+  if (n < (uint64_t(1) << 32)) {
+    return jeaiii_9_or_10(b, n);
+  }
+  // At least 10 digits: the low 8 digits, and 2 to 12 digits above them.
+  const uint32_t z = uint32_t(n % 100000000);
+  uint64_t u = n / 100000000;
+  if (u < 100000000) {
+    // u has 2 to 8 digits (if u < 10, n would be below 2^32).
+    b = jeaiii_lt1e8(b, uint32_t(u));
+  } else if (u < (uint64_t(1) << 32)) {
+    b = jeaiii_9_or_10(b, u);
+  } else {
+    // u has 11 or 12 digits: split off 8 more.
+    const uint32_t y = uint32_t(u % 100000000);
+    u /= 100000000;
+    b = jeaiii_lt1e8(b, uint32_t(u)); // 3 or 4 digits
+    b = jeaiii_8_digits(b, y);
   }
-  else {
-    return fast_digit_count_64(static_cast<uint64_t>(v));
-  }
-}
-static const char decimal_table[200] = {
-    0x30, 0x30, 0x30, 0x31, 0x30, 0x32, 0x30, 0x33, 0x30, 0x34, 0x30, 0x35,
-    0x30, 0x36, 0x30, 0x37, 0x30, 0x38, 0x30, 0x39, 0x31, 0x30, 0x31, 0x31,
-    0x31, 0x32, 0x31, 0x33, 0x31, 0x34, 0x31, 0x35, 0x31, 0x36, 0x31, 0x37,
-    0x31, 0x38, 0x31, 0x39, 0x32, 0x30, 0x32, 0x31, 0x32, 0x32, 0x32, 0x33,
-    0x32, 0x34, 0x32, 0x35, 0x32, 0x36, 0x32, 0x37, 0x32, 0x38, 0x32, 0x39,
-    0x33, 0x30, 0x33, 0x31, 0x33, 0x32, 0x33, 0x33, 0x33, 0x34, 0x33, 0x35,
-    0x33, 0x36, 0x33, 0x37, 0x33, 0x38, 0x33, 0x39, 0x34, 0x30, 0x34, 0x31,
-    0x34, 0x32, 0x34, 0x33, 0x34, 0x34, 0x34, 0x35, 0x34, 0x36, 0x34, 0x37,
-    0x34, 0x38, 0x34, 0x39, 0x35, 0x30, 0x35, 0x31, 0x35, 0x32, 0x35, 0x33,
-    0x35, 0x34, 0x35, 0x35, 0x35, 0x36, 0x35, 0x37, 0x35, 0x38, 0x35, 0x39,
-    0x36, 0x30, 0x36, 0x31, 0x36, 0x32, 0x36, 0x33, 0x36, 0x34, 0x36, 0x35,
-    0x36, 0x36, 0x36, 0x37, 0x36, 0x38, 0x36, 0x39, 0x37, 0x30, 0x37, 0x31,
-    0x37, 0x32, 0x37, 0x33, 0x37, 0x34, 0x37, 0x35, 0x37, 0x36, 0x37, 0x37,
-    0x37, 0x38, 0x37, 0x39, 0x38, 0x30, 0x38, 0x31, 0x38, 0x32, 0x38, 0x33,
-    0x38, 0x34, 0x38, 0x35, 0x38, 0x36, 0x38, 0x37, 0x38, 0x38, 0x38, 0x39,
-    0x39, 0x30, 0x39, 0x31, 0x39, 0x32, 0x39, 0x33, 0x39, 0x34, 0x39, 0x35,
-    0x39, 0x36, 0x39, 0x37, 0x39, 0x38, 0x39, 0x39,
-};
+  return jeaiii_8_digits(b, z);
+}
+
+// Writes v at p, which must have to_chars_buffer_size bytes available, and
+// returns the end of what was written.
+simdjson_inline char *write_double(char *p, double v) noexcept {
+#if SIMDJSON_ENABLE_NAN_INF
+  if (simdjson_unlikely(!std::isfinite(v))) {
+    if (std::isnan(v)) {
+      std::memcpy(p, "NaN", 3);
+      return p + 3;
+    }
+    if (v < 0) {
+      *p++ = '-';
+    }
+    std::memcpy(p, "Infinity", 8);
+    return p + 8;
+  }
+#endif
+  return simdjson::internal::to_chars(p, nullptr, v);
+}
 } // namespace internal

 template <typename number_type, typename>
@@ -53510,87 +65270,62 @@ simdjson_inline void string_builder::append(number_type v) noexcept {
     }
   }
   else SIMDJSON_IF_CONSTEXPR(std::is_unsigned<number_type>::value) {
-    // Process 4 digits at a time instead of 2, reducing store operations
-    // and divisions by approximately half for large numbers.
     constexpr size_t max_number_size = 20;
     if (capacity_check(max_number_size)) {
       using unsigned_type = typename std::make_unsigned<number_type>::type;
-      unsigned_type pv = static_cast<unsigned_type>(v);
-      size_t dc = internal::digit_count(pv);
-      char *write_pointer = buffer.get() + position + dc - 1;
-
-      // Process 4 digits per iteration for large numbers
-      while (pv >= 10000) {
-        unsigned_type q = pv / 10000;
-        unsigned_type r = pv % 10000;
-        unsigned_type r_hi = r / 100;  // High 2 digits of remainder
-        unsigned_type r_lo = r % 100;  // Low 2 digits of remainder
-        // Write low 2 digits first (rightmost), then high 2 digits
-        memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
-        memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
-        write_pointer -= 4;
-        pv = q;
-      }
-
-      // Handle remaining 1-4 digits with original 2-digit loop
-      while (pv >= 100) {
-        memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
-        write_pointer -= 2;
-        pv /= 100;
-      }
-      if (pv >= 10) {
-        *write_pointer-- = char('0' + (pv % 10));
-        pv /= 10;
-      }
-      *write_pointer = char('0' + pv);
-      position += dc;
+      char* end = internal::write_uint_jeaiii(
+          buffer.get() + position,
+          static_cast<uint64_t>(static_cast<unsigned_type>(v)));
+      position = end - buffer.get();
     }
   }
   else SIMDJSON_IF_CONSTEXPR(std::is_integral<number_type>::value) {
-    // Same 4-digit batching as unsigned path for signed integers
+    // 19 digits (max abs value of int64_t) + optional minus sign.
     constexpr size_t max_number_size = 20;
     if (capacity_check(max_number_size)) {
       using unsigned_type = typename std::make_unsigned<number_type>::type;
       bool negative = v < 0;
-      unsigned_type pv = static_cast<unsigned_type>(v);
-      if (negative) {
-        pv = 0 - pv; // the 0 is for Microsoft
-      }
-      size_t dc = internal::digit_count(pv);
-      // by always writing the minus sign, we avoid the branch.
+      // 0 - pv (rather than -pv) avoids an MSVC unary-minus warning.
+      unsigned_type pv = negative
+          ? unsigned_type(0) - static_cast<unsigned_type>(v)
+          : static_cast<unsigned_type>(v);
+      // Branchless: always write '-', advance only if negative.
       buffer.get()[position] = '-';
-      position += negative ? 1 : 0;
-      char *write_pointer = buffer.get() + position + dc - 1;
-
-      // Process 4 digits per iteration for large numbers
-      while (pv >= 10000) {
-        unsigned_type q = pv / 10000;
-        unsigned_type r = pv % 10000;
-        unsigned_type r_hi = r / 100;
-        unsigned_type r_lo = r % 100;
-        memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
-        memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
-        write_pointer -= 4;
-        pv = q;
-      }
-
-      // Handle remaining 1-4 digits
-      while (pv >= 100) {
-        memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
-        write_pointer -= 2;
-        pv /= 100;
-      }
-      if (pv >= 10) {
-        *write_pointer-- = char('0' + (pv % 10));
-        pv /= 10;
-      }
-      *write_pointer = char('0' + pv);
-      position += dc;
+      position += negative;
+      char* end = internal::write_uint_jeaiii(
+          buffer.get() + position, static_cast<uint64_t>(pv));
+      position = end - buffer.get();
     }
   }
   else SIMDJSON_IF_CONSTEXPR(std::is_floating_point<number_type>::value) {
-    constexpr size_t max_number_size = 24;
+    // Must reserve to_chars_buffer_size (40): only ~24 chars are emitted,
+    // but to_chars over-writes with fixed-size 16/17-byte copies so the
+    // compiler can inline mem* (see simdjson::internal::to_chars_buffer_size).
+    constexpr size_t max_number_size = simdjson::internal::to_chars_buffer_size;
     if (capacity_check(max_number_size)) {
+#if SIMDJSON_ENABLE_NAN_INF
+      // Check if the input might be NaN or infinity
+      if (simdjson_unlikely(!std::isfinite(v))) {
+        if (std::isnan(v)) {
+          constexpr char nan_literal[] = "NaN";
+          constexpr size_t nan_len = sizeof(nan_literal) - 1;
+
+          std::memcpy(buffer.get() + position, nan_literal, nan_len);
+          position += nan_len;
+        } else {
+          constexpr char inf_literal[] = "Infinity";
+          constexpr size_t inf_len = sizeof(inf_literal) - 1;
+          if (v < 0) {
+            buffer.get()[position] = '-';
+            ++position;
+          }
+          std::memcpy(buffer.get() + position, inf_literal, inf_len);
+          position += inf_len;
+        }
+        return;
+      }
+#endif
+
       // We could specialize for float.
       char *end = simdjson::internal::to_chars(buffer.get() + position, nullptr,
                                                double(v));
@@ -53651,7 +65386,9 @@ simdjson_inline void string_builder::escape_and_append_with_quotes() noexcept {
 #endif

 simdjson_inline void string_builder::append_raw(const char *c) noexcept {
-  size_t len = std::strlen(c);
+  // char_traits::length is constexpr; lets the compiler fold the length
+  // when called with a pointer to a compile-time-constant string.
+  size_t len = std::char_traits<char>::length(c);
   append_raw(c, len);
 }

@@ -53670,6 +65407,14 @@ simdjson_inline void string_builder::append_raw(const char *str,
     position += len;
   }
 }
+
+template <size_t N>
+simdjson_inline void string_builder::append_raw_n(const char *str) noexcept {
+  if (capacity_check(N)) {
+    std::memcpy(buffer.get() + position, str, N);
+    position += N;
+  }
+}
 #if SIMDJSON_SUPPORTS_CONCEPTS
 // Support for optional types (std::optional, etc.)
 template <concepts::optional_type T>
@@ -53699,7 +65444,7 @@ simdjson_inline void string_builder::append(const T &value) {
 #if SIMDJSON_SUPPORTS_RANGES && SIMDJSON_SUPPORTS_CONCEPTS
 // Support for range-based appending (std::ranges::view, etc.)
 template <std::ranges::range R>
-  requires(!std::is_convertible<R, std::string_view>::value && !require_custom_serialization<R>)
+  requires(!std::is_convertible<R, std::string_view>::value && !concepts::optional_type<R> && !require_custom_serialization<R>)
 simdjson_inline void string_builder::append(const R &range) noexcept {
   auto it = std::ranges::begin(range);
   auto end = std::ranges::end(range);
@@ -53980,10 +65725,6 @@ simdjson_inline int count_ones(uint64_t input_num) {
   return __lsx_vpickve2gr_w(__lsx_vpcnt_d(__m128i(v2u64{input_num, 0})), 0);
 }

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2, uint64_t *result) {
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-}

 } // unnamed namespace
 } // namespace lsx
@@ -54698,7 +66439,7 @@ public:
 #if SIMDJSON_SUPPORTS_RANGES && SIMDJSON_SUPPORTS_CONCEPTS
   // Support for range-based appending (std::ranges::view, etc.)
   template <std::ranges::range R>
-requires (!std::is_convertible<R, std::string_view>::value && !require_custom_serialization<R>)
+requires (!std::is_convertible<R, std::string_view>::value && !concepts::optional_type<R> && !require_custom_serialization<R>)
   simdjson_inline void append(const R &range) noexcept;
 #endif
   /**
@@ -54712,6 +66453,15 @@ requires (!std::is_convertible<R, std::string_view>::value && !require_custom_se
    * There is no UTF-8 validation.
    */
   simdjson_inline void append_raw(const char *str, size_t len) noexcept;
+
+  /**
+   * Append exactly N characters from str. The length is a template parameter
+   * so the compiler can fully inline the memcpy with a compile-time-constant
+   * size, avoiding the libc call. Used for compile-time-constant keys in the
+   * reflection struct atom.
+   */
+  template <size_t N>
+  simdjson_inline void append_raw_n(const char *str) noexcept;
 #if SIMDJSON_EXCEPTIONS
   /**
    * Creates an std::string from the written JSON buffer.
@@ -54763,6 +66513,26 @@ requires (!std::is_convertible<R, std::string_view>::value && !require_custom_se
    */
   simdjson_inline size_t size() const noexcept;

+  // ============================================================
+  // Internal hooks for the position-as-local writer in json_builder.h.
+  // These exist so the reflection atom code can hold buffer pointer,
+  // position and capacity in registers across long write chains rather
+  // than reloading them after every char* write (strict aliasing
+  // forces those reloads when accessed via members of *this). User
+  // code should NOT call these directly.
+  // ============================================================
+  simdjson_inline char *unsafe_data() noexcept { return buffer.get(); }
+  simdjson_inline size_t unsafe_position() const noexcept { return position; }
+  simdjson_inline size_t unsafe_capacity() const noexcept { return capacity; }
+  simdjson_inline void unsafe_set_position(size_t p) noexcept { position = p; }
+  /// Make capacity available for at least `n` more bytes after the current
+  /// position. Returns false if the allocation failed.
+  simdjson_inline bool unsafe_grow(size_t needed_total_capacity) noexcept {
+    grow_buffer(needed_total_capacity);
+    return is_valid;
+  }
+  simdjson_inline bool unsafe_is_valid() const noexcept { return is_valid; }
+
 private:
   /**
    * Returns true if we can write at least upcoming_bytes bytes.
@@ -54776,7 +66546,7 @@ private:
    * If the allocation fails, is_valid is set to false. We expect
    * that this function would not be repeatedly called.
    */
-  simdjson_inline void grow_buffer(size_t desired_capacity);
+  inline void grow_buffer(size_t desired_capacity);

   /**
    * We use this helper function to make sure that is_valid is kept consistent.
@@ -54833,6 +66603,7 @@ simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initi
 /* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_STRING_BUILDER_H */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/builder/json_string_builder.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/concepts.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/annotations.h" */
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
 #if SIMDJSON_STATIC_REFLECTION

@@ -54850,64 +66621,370 @@ namespace simdjson {
 namespace lsx {
 namespace builder {

-template <class T>
+// Forward-declare helpers defined in json_string_builder-inl.h so the
+// writer-based atom code below can call them (the -inl.h is not yet
+// included at the point this header is parsed; without these forwards,
+// name lookup falls back to the wrong outer namespace).
+namespace internal {
+simdjson_really_inline char *write_uint_jeaiii(char *p, uint64_t v) noexcept;
+simdjson_inline char *write_double(char *p, double v) noexcept;
+} // namespace internal
+simdjson_really_inline size_t write_string_escaped(const std::string_view input, char *out);
+
+SIMDJSON_PUSH_DISABLE_WARNINGS
+SIMDJSON_DISABLE_GCC_WARNING(-Warray-bounds)
+#if !defined(__clang__)
+SIMDJSON_DISABLE_GCC_WARNING(-Wstringop-overflow)
+#endif
+
+// =============================================================
+// `writer`: position-as-local hot-path writer used by the reflection
+// atom code below. Holds the buffer pointer, write position and
+// capacity in three fields that, once `writer` itself is a stack-local
+// in the caller and all atom() functions are inlined, become true
+// register-resident locals after SROA. Glaze achieves the same effect
+// by passing `B&& b, auto&& ix` through every helper. Holding `pos`
+// in a register (rather than as a member of string_builder) is what
+// breaks the strict-aliasing penalty on every char* write through the
+// buffer, which forces a reload of `b.position` and `b.capacity`
+// after every byte.
+//
+// basic_writer<false> (below) is the unchecked variant: the caller has
+// already reserved enough capacity for everything the write chain can
+// produce (see bound_detail::size_bound), so ensure() compiles away.
+// =============================================================
+template <bool Checked>
+struct basic_writer {
+  static constexpr bool checked = Checked;
+  char *ptr;        // buffer pointer (refreshed after a grow)
+  size_t pos;       // write position (local)
+  size_t cap;       // capacity (refreshed after a grow)
+  string_builder &sb;  // back-ref for grow / sync
+
+  // Snapshot string_builder state into a writer for the duration of
+  // a write chain.
+  simdjson_really_inline basic_writer(string_builder &builder) noexcept
+      : ptr(builder.unsafe_data())
+      , pos(builder.unsafe_position())
+      , cap(builder.unsafe_capacity())
+      , sb(builder) {}
+
+  // Write the local position back to the underlying string_builder.
+  // Caller is responsible for invoking before the writer is dropped
+  // (otherwise data is lost). Idempotent.
+  simdjson_really_inline void sync() noexcept {
+    sb.unsafe_set_position(pos);
+  }
+
+  // Ensure at least `n` more bytes of free capacity. Grows the
+  // underlying buffer if needed (rare path). Returns false on
+  // allocation failure.
+  simdjson_really_inline bool ensure(size_t n) noexcept {
+    // pos <= cap, and cap is the size of a live allocation, so pos + n
+    // cannot wrap when n is a small constant or a compile-time length.
+    // Callers passing a size derived from input (the string atoms) must
+    // bound it against pos themselves. Keep the `pos + n <= cap` form:
+    // `n <= cap - pos` is measurably slower once the serializer is inlined.
+    if (simdjson_likely(pos + n <= cap)) { return true; }
+    return grow_slow(n);
+  }
+
+  simdjson_never_inline bool grow_slow(size_t n) noexcept {
+    // Detect overflow.
+    // This is pedantic except maybe on 32-bit targets.
+    if (simdjson_unlikely(pos + n < pos)) return false;
+    sb.unsafe_set_position(pos);
+    // even if 2*capacity overflows, the (std::max) below will pick the needed value,
+    // so we do not need a separate overflow check here.
+    if (!sb.unsafe_grow((std::max)(cap * 2, pos + n))) {
+      // The string_builder freed its buffer and is now invalid (null buffer,
+      // zero capacity and position). Mirror that state so that every later
+      // ensure() fails too: callers only return from the current atom, and
+      // their callers keep writing.
+      ptr = nullptr;
+      pos = 0;
+      cap = 0;
+      return false;
+    }
+    ptr = sb.unsafe_data();
+    cap = sb.unsafe_capacity();
+    return true;
+  }
+};
+
+// The unchecked writer writes into a raw buffer that the caller sized with
+// serialized_size_bound: it never grows and needs no string_builder.
+template <>
+struct basic_writer<false> {
+  static constexpr bool checked = false;
+  char *ptr;
+  size_t pos;
+
+  simdjson_really_inline basic_writer(char *buffer, size_t position) noexcept
+      : ptr(buffer), pos(position) {}
+
+  simdjson_really_inline bool ensure(size_t) const noexcept { return true; }
+};
+
+using writer = basic_writer<true>;
+using unchecked_writer = basic_writer<false>;
+
+// Bytes reserved past the size bound for an unchecked writer: it may then
+// write a little past the end of what it produces (e.g., copy keys as whole
+// 16-byte blocks).
+inline constexpr size_t unchecked_slack = 64;
+
+consteval size_t padded_key_length(size_t length) {
+  return (length + 15) / 16 * 16;
+}
+
+// === Helper: invoke a string_builder member that writes variable-length
+// content (escape_and_append_with_quotes etc), syncing the writer's local
+// state before the call and reloading after. Used for string fields where
+// rewriting the entire SIMD escape path through the writer would be a much
+// bigger refactor. f may be user code (a with<Adapter> serializer) that
+// throws: the exception then propagates to the caller.
+template <class W, class F>
+simdjson_really_inline void call_through_string_builder(W &w, F &&f) noexcept(noexcept(f(w.sb))) {
+  w.sync();
+  f(w.sb);
+  w.ptr = w.sb.unsafe_data();
+  w.pos = w.sb.unsafe_position();
+  w.cap = w.sb.unsafe_capacity();
+}
+
+// Helpers implementing the serialization side of the annotations (see
+// simdjson/annotations.h) for reflected structures.
+namespace annotation_detail {
+
+// A member is serialized unless it is annotated with skip or skip_serializing.
+consteval bool is_serialized_member(std::meta::info dm) {
+  return !simdjson::detail::has_annotation(dm, ^^simdjson::detail::skip_tag)
+      && !simdjson::detail::has_annotation(dm, ^^simdjson::detail::skip_serializing_tag);
+}
+
+// False when the member has a skip_serializing_if<pred> annotation and
+// pred(value) is true.
+template <auto dm, typename V>
+simdjson_really_inline bool should_serialize(const V &value) {
+  constexpr std::meta::info skip_if_type = simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::skip_serializing_if_t);
+  if constexpr (skip_if_type != std::meta::info{}) {
+    using skip_if = typename [: skip_if_type :];
+    return !skip_if::predicate(value);
+  } else {
+    (void)value;
+    return true;
+  }
+}
+
+// Serialize a member value, through its with<Adapter> annotation when the
+// adapter provides a serialize function.
+template <auto dm, class W, typename V>
+simdjson_really_inline void atom_member(W &w, const V &value) {
+  constexpr std::meta::info with_type = simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::with_t);
+  if constexpr (with_type != std::meta::info{}) {
+    using adapter = typename [: with_type :]::adapter;
+    if constexpr (requires(string_builder &b) { adapter::serialize(b, value); }) {
+      call_through_string_builder(w, [&](string_builder &b) { adapter::serialize(b, value); });
+    } else {
+      atom(w, value);
+    }
+  } else {
+    atom(w, value);
+  }
+}
+
+// Write the "key":value pairs of the members of t (without the braces), each
+// preceded by a comma unless it is the first one. The members of a member
+// annotated with flatten are written in its place.
+template <class W, class T>
+simdjson_really_inline void atom_fields(W &w, const T &t, bool &first) {
+  // Per-field block: ensure key+value worst case, then write key + value
+  // through the writer's local pos. For arithmetic fields, the integer
+  // write happens directly via write_uint_jeaiii on w.ptr+w.pos, so pos
+  // never round-trips through memory.
+  template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+    if constexpr (is_serialized_member(dm)) {
+      if (should_serialize<dm>(t.[:dm:])) {
+        if constexpr (simdjson::detail::has_annotation(dm, ^^simdjson::detail::flatten_tag)) {
+          static_assert(std::meta::is_class_type(simdjson::detail::flattened_type(dm)));
+          using flattened = std::remove_cvref_t<decltype(t.[:dm:])>;
+          static_assert(!concepts::container_but_not_string<flattened> && !concepts::string_view_keyed_map<flattened> &&
+                        !concepts::appendable_containers<flattened> && !concepts::optional_type<flattened> &&
+                        !concepts::smart_pointer<flattened> && !std::is_same_v<flattened, std::string> &&
+                        !std::is_same_v<flattened, std::string_view> && !require_custom_serialization<flattened>,
+                        "simdjson::flatten requires a member whose type is a structure serialized member by member");
+          atom_fields(w, t.[:dm:], first);
+        } else {
+          // Copy the key as whole 16-byte blocks from a zero-padded copy (one
+          // load and one store); ensure() reserves the padded length, and the
+          // unchecked writer has slack past its bound. Prior related work:
+          // jsonifier copies a power-of-two padded key and advances the cursor
+          // by the real length (serialize_impl.hpp, packed_blitter,
+          // https://github.com/nihilai-collective/Jsonifier).
+          constexpr const char* key_name = simdjson::get_json_key_name<dm>();
+          constexpr size_t first_key_len = constevalutil::consteval_to_quoted_escaped(key_name).size() + 1;
+          constexpr size_t rest_key_len = first_key_len + 1;
+          constexpr auto first_key = std::define_static_string(
+              constevalutil::consteval_to_quoted_escaped(key_name) + ":" +
+              std::string(padded_key_length(first_key_len) - first_key_len, '\0'));
+          constexpr auto rest_key = std::define_static_string(
+              std::string(",") + constevalutil::consteval_to_quoted_escaped(key_name) + ":" +
+              std::string(padded_key_length(rest_key_len) - rest_key_len, '\0'));
+          if (!w.ensure(padded_key_length(rest_key_len))) { return; }
+          if (first) {
+            std::memcpy(w.ptr + w.pos, first_key, padded_key_length(first_key_len));
+            w.pos += first_key_len;
+          } else {
+            std::memcpy(w.ptr + w.pos, rest_key, padded_key_length(rest_key_len));
+            w.pos += rest_key_len;
+          }
+          first = false;
+          atom_member<dm>(w, t.[:dm:]);
+        }
+      }
+    }
+  };
+}
+
+} // namespace annotation_detail
+
+template <class W, class T>
   requires(concepts::container_but_not_string<T> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &t) {
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
   auto it = t.begin();
   auto end = t.end();
   if (it == end) {
-    b.append_raw("[]");
+    if (!w.ensure(2)) return;
+    std::memcpy(w.ptr + w.pos, "[]", 2);
+    w.pos += 2;
     return;
   }
-  b.append('[');
-  atom(b, *it);
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = '[';
+  atom(w, *it);
   ++it;
   for (; it != end; ++it) {
-    b.append(',');
-    atom(b, *it);
+    if (!w.ensure(1)) return;
+    w.ptr[w.pos++] = ',';
+    atom(w, *it);
   }
-  b.append(']');
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = ']';
 }

-template <class T>
+template <class W, class T>
   requires(std::is_same_v<T, std::string> ||
            std::is_same_v<T, std::string_view> ||
            std::is_same_v<T, const char *> ||
            std::is_same_v<T, char>)
-constexpr void atom(string_builder &b, const T &t) {
-  b.escape_and_append_with_quotes(t);
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
+  // Inline the escape path through the writer so we never round-trip
+  // pos through memory for string fields (Twitter is dominated by
+  // these -- sync/reload around each string was a real cost).
+  std::string_view input;
+  if constexpr (std::is_same_v<T, char>) {
+    input = std::string_view(&t, 1);
+  } else {
+    input = std::string_view(t);
+  }
+  // Worst-case escape: every byte expands to \uXXXX (6 chars), plus 2 quotes.
+  // Guard against w.pos + 2 + 6 * input.size() wrapping for huge inputs -- if
+  // it wrapped to a small value, ensure() would spuriously succeed and the
+  // subsequent escape would overflow the buffer. max - w.pos cannot wrap, and
+  // size < (max - pos) / 6 implies pos + 6 * size + 6 <= max.
+  // Note that this is pedantic except maybe on 32-bit targets.
+  if constexpr (W::checked) {
+    if (simdjson_unlikely(input.size() >= ((std::numeric_limits<size_t>::max)() - w.pos) / 6)) { return; }
+    if (!w.ensure(2 + 6 * input.size())) { return; }
+  }
+  w.ptr[w.pos++] = '"';
+  w.pos += write_string_escaped(input, w.ptr + w.pos);
+  w.ptr[w.pos++] = '"';
 }

-template <concepts::string_view_keyed_map T>
+template <class W, concepts::string_view_keyed_map T>
   requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &m) {
+simdjson_really_inline constexpr void atom(W &w, const T &m) {
   if (m.empty()) {
-    b.append_raw("{}");
+    if (!w.ensure(2)) return;
+    std::memcpy(w.ptr + w.pos, "{}", 2);
+    w.pos += 2;
     return;
   }
-  b.append('{');
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = '{';
   bool first = true;
   for (const auto& [key, value] : m) {
     if (!first) {
-      b.append(',');
+      if (!w.ensure(1)) return;
+      w.ptr[w.pos++] = ',';
     }
     first = false;
-    // Keys must be convertible to string_view per the concept
-    b.escape_and_append_with_quotes(key);
-    b.append(':');
-    atom(b, value);
+    // Keys must be convertible to string_view per the concept.
+    std::string_view key_sv(key);
+    // Guard against w.pos + 3 + 6 * key_sv.size() wrapping for huge keys, if
+    // it wrapped to a small value, ensure() would spuriously succeed and the
+    // subsequent escape would overflow the buffer. max - w.pos cannot wrap, and
+    // size < (max - pos) / 6 implies pos + 6 * size + 6 <= max.
+    // Note that this is pedantic except maybe on 32-bit targets.
+    if constexpr (W::checked) {
+      if (simdjson_unlikely(key_sv.size() >= ((std::numeric_limits<size_t>::max)() - w.pos) / 6)) { return; }
+      if (!w.ensure(2 + 6 * key_sv.size() + 1)) { return; }
+    }
+    w.ptr[w.pos++] = '"';
+    w.pos += write_string_escaped(key_sv, w.ptr + w.pos);
+    w.ptr[w.pos++] = '"';
+    w.ptr[w.pos++] = ':';
+    atom(w, value);
   }
-  b.append('}');
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = '}';
 }


-template<typename number_type,
+template<class W, typename number_type,
          typename = typename std::enable_if<std::is_arithmetic<number_type>::value && !std::is_same_v<number_type, char>>::type>
-constexpr void atom(string_builder &b, const number_type t) {
-  b.append(t);
+simdjson_really_inline constexpr void atom(W &w, const number_type t) {
+  // Booleans / floats: defer to string_builder (rare path; keeps writer hot
+  // path free of float-formatter machinery). For integers, write directly
+  // via jeaiii using local pos.
+  if constexpr (std::is_same_v<number_type, bool>) {
+    if (t) {
+      if (!w.ensure(4)) return;
+      std::memcpy(w.ptr + w.pos, "true", 4);
+      w.pos += 4;
+    } else {
+      if (!w.ensure(5)) return;
+      std::memcpy(w.ptr + w.pos, "false", 5);
+      w.pos += 5;
+    }
+  } else if constexpr (std::is_floating_point_v<number_type>) {
+    if constexpr (W::checked) {
+      call_through_string_builder(w, [&](string_builder &b) { b.append(t); });
+    } else {
+      w.pos = size_t(internal::write_double(w.ptr + w.pos, double(t)) - w.ptr);
+    }
+  } else if constexpr (std::is_unsigned_v<number_type>) {
+    if (!w.ensure(20)) return;
+    char *end = internal::write_uint_jeaiii(
+        w.ptr + w.pos, static_cast<uint64_t>(t));
+    w.pos = static_cast<size_t>(end - w.ptr);
+  } else {
+    // signed integral
+    if (!w.ensure(20)) return;
+    using U = typename std::make_unsigned<number_type>::type;
+    bool negative = t < 0;
+    U pv = negative ? U(0) - static_cast<U>(t) : static_cast<U>(t);
+    w.ptr[w.pos] = '-';
+    w.pos += negative;
+    char *end = internal::write_uint_jeaiii(
+        w.ptr + w.pos, static_cast<uint64_t>(pv));
+    w.pos = static_cast<size_t>(end - w.ptr);
+  }
 }

-template <class T>
+template <class W, class T>
   requires(std::is_class_v<T> && !concepts::container_but_not_string<T> &&
            !concepts::string_view_keyed_map<T> &&
            !concepts::optional_type<T> &&
@@ -54917,92 +66994,259 @@ template <class T>
            !std::is_same_v<T, std::string_view> &&
            !std::is_same_v<T, const char*> &&
            !std::is_same_v<T, char> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &t) {
-  int i = 0;
-  b.append('{');
-  template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
-    if (i != 0)
-      b.append(',');
-    constexpr auto key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(dm)));
-    b.append_raw(key);
-    b.append(':');
-    atom(b, t.[:dm:]);
-    i++;
-  };
-  b.append('}');
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
+  if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag)) {
+    // A transparent structure is serialized as its single member.
+    constexpr auto dm = simdjson::detail::transparent_member(^^T);
+    annotation_detail::atom_member<dm>(w, t.[:dm:]);
+  } else {
+    bool first = true;
+    if (!w.ensure(1)) { return; }
+    w.ptr[w.pos++] = '{';
+    annotation_detail::atom_fields(w, t, first);
+    if (!w.ensure(1)) { return; }
+    w.ptr[w.pos++] = '}';
+  }
 }

 // Support for optional types (std::optional, etc.)
-template <concepts::optional_type T>
+template <class W, concepts::optional_type T>
   requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &opt) {
+simdjson_really_inline constexpr void atom(W &w, const T &opt) {
   if (opt) {
-    atom(b, opt.value());
+    atom(w, opt.value());
   } else {
-    b.append_raw("null");
+    if (!w.ensure(4)) return;
+    std::memcpy(w.ptr + w.pos, "null", 4);
+    w.pos += 4;
   }
 }

 // Support for smart pointers (std::unique_ptr, std::shared_ptr, etc.)
-template <concepts::smart_pointer T>
+template <class W, concepts::smart_pointer T>
   requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &ptr) {
+simdjson_really_inline constexpr void atom(W &w, const T &ptr) {
   if (ptr) {
-    atom(b, *ptr);
+    atom(w, *ptr);
   } else {
-    b.append_raw("null");
+    if (!w.ensure(4)) return;
+    std::memcpy(w.ptr + w.pos, "null", 4);
+    w.pos += 4;
   }
 }

 // Support for enums - serialize as string representation using expand approach from P2996R12
-template <typename T>
+template <class W, typename T>
   requires(std::is_enum_v<T> && !require_custom_serialization<T>)
-void atom(string_builder &b, const T &e) {
+simdjson_really_inline void atom(W &w, const T &e) {
 #if SIMDJSON_STATIC_REFLECTION
-  constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
+  static constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
   template for (constexpr auto enum_val : enumerators) {
-    constexpr auto enum_str = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(enum_val)));
+    constexpr auto enum_str = std::define_static_string(constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<enum_val>()));
+    constexpr size_t enum_str_len = std::char_traits<char>::length(enum_str);
     if (e == [:enum_val:]) {
-      b.append_raw(enum_str);
+      if (!w.ensure(enum_str_len)) return;
+      std::memcpy(w.ptr + w.pos, enum_str, enum_str_len);
+      w.pos += enum_str_len;
       return;
     }
   };
   // Fallback to integer if enum value not found
-  atom(b, static_cast<std::underlying_type_t<T>>(e));
+  atom(w, static_cast<std::underlying_type_t<T>>(e));
 #else
   // Fallback: serialize as integer if reflection not available
-  atom(b, static_cast<std::underlying_type_t<T>>(e));
+  atom(w, static_cast<std::underlying_type_t<T>>(e));
 #endif
 }

 // Support for appendable containers that don't have operator[] (sets, etc.)
-template <concepts::appendable_containers T>
+template <class W, concepts::appendable_containers T>
   requires(!concepts::container_but_not_string<T> && !concepts::string_view_keyed_map<T> &&
            !concepts::optional_type<T> && !concepts::smart_pointer<T> &&
            !std::is_same_v<T, std::string> &&
            !std::is_same_v<T, std::string_view> && !std::is_same_v<T, const char*> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &container) {
+simdjson_really_inline constexpr void atom(W &w, const T &container) {
   if (container.empty()) {
-    b.append_raw("[]");
+    if (!w.ensure(2)) return;
+    std::memcpy(w.ptr + w.pos, "[]", 2);
+    w.pos += 2;
     return;
   }
-  b.append('[');
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = '[';
   bool first = true;
   for (const auto& item : container) {
     if (!first) {
-      b.append(',');
+      if (!w.ensure(1)) return;
+      w.ptr[w.pos++] = ',';
     }
     first = false;
-    atom(b, item);
+    atom(w, item);
+  }
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = ']';
+}
+
+// =============================================================
+// Size bound: an upper bound on the number of bytes that atom(w, t) writes.
+// Computing it first lets append() reserve the capacity once and then run
+// the whole write chain through an unchecked_writer, without a capacity
+// check before every write. It mirrors the atom() overloads above.
+//
+// Prior related work: jsonifier sizes the document first, counting 6 bytes
+// per string byte, resizes once to that bound plus slack, and writes with
+// no capacity check on each store. to_json() does that resize through
+// std::string::resize_and_overwrite. See serializer.hpp and
+// serialize_impl.hpp in https://github.com/nihilai-collective/Jsonifier
+// and https://nihilai-collective.net/serialization.
+// =============================================================
+namespace bound_detail {
+
+// Whether size_bound covers everything that atom() writes for T: not when a
+// member is serialized by a with<Adapter> serializer, which writes an unknown
+// amount through the string_builder.
+template <class T>
+consteval bool is_bounded() {
+  if constexpr (require_custom_serialization<T>) {
+    return false;
+  } else if constexpr (std::is_same_v<T, std::string> || std::is_same_v<T, std::string_view> ||
+                       std::is_same_v<T, const char *> || std::is_arithmetic_v<T> || std::is_enum_v<T>) {
+    return true;
+  } else if constexpr (concepts::optional_type<T> || concepts::smart_pointer<T>) {
+    return is_bounded<std::remove_cvref_t<decltype(*std::declval<const T &>())>>();
+  } else if constexpr (concepts::string_view_keyed_map<T>) {
+    return is_bounded<std::remove_cvref_t<typename T::mapped_type>>();
+  } else if constexpr (concepts::container_but_not_string<T> || concepts::appendable_containers<T>) {
+    return is_bounded<std::remove_cvref_t<std::ranges::range_value_t<T>>>();
+  } else {
+    bool bounded = true;
+    template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+      if constexpr (annotation_detail::is_serialized_member(dm)) {
+        bounded = bounded && simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::with_t) == std::meta::info{} &&
+                  is_bounded<std::remove_cvref_t<decltype(std::declval<const T &>().[:dm:])>>();
+      }
+    };
+    return bounded;
+  }
+}
+
+template <class T>
+consteval size_t enum_bound() {
+  size_t bound = 20; // the integer fallback
+  template for (constexpr auto enum_val : std::define_static_array(std::meta::enumerators_of(^^T))) {
+    constexpr size_t len = std::char_traits<char>::length(std::define_static_string(
+        constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<enum_val>())));
+    bound = (std::max)(bound, len);
+  };
+  return bound;
+}
+
+template <class T>
+simdjson_really_inline size_t size_bound(const T &t) noexcept;
+
+// Bound for the "key":value pairs of a structure, commas included.
+template <class T>
+simdjson_really_inline size_t fields_bound(const T &t) noexcept {
+  size_t bound = 0;
+  template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+    if constexpr (annotation_detail::is_serialized_member(dm)) {
+      if constexpr (simdjson::detail::has_annotation(dm, ^^simdjson::detail::flatten_tag)) {
+        bound += fields_bound(t.[:dm:]);
+      } else {
+        constexpr size_t rest_key_len = constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<dm>()).size() + 2;
+        bound += rest_key_len + size_bound(t.[:dm:]);
+      }
+    }
+  };
+  return bound;
+}
+
+template <class T>
+simdjson_really_inline size_t size_bound([[maybe_unused]] const T &t) noexcept {
+  if constexpr (std::is_same_v<T, char>) {
+    return 2 + 6;
+  } else if constexpr (std::is_same_v<T, std::string> || std::is_same_v<T, std::string_view> ||
+                       std::is_same_v<T, const char *>) {
+    // Every byte may become \uXXXX, plus the quotes.
+    return 2 + 6 * std::string_view(t).size();
+  } else if constexpr (std::is_same_v<T, bool>) {
+    return 5;
+  } else if constexpr (std::is_floating_point_v<T>) {
+    return simdjson::internal::to_chars_buffer_size;
+  } else if constexpr (std::is_arithmetic_v<T>) {
+    return 20;
+  } else if constexpr (std::is_enum_v<T>) {
+    return enum_bound<T>();
+  } else if constexpr (concepts::optional_type<T> || concepts::smart_pointer<T>) {
+    return t ? size_bound(*t) : 4;
+  } else if constexpr (concepts::string_view_keyed_map<T>) {
+    size_t bound = 2;
+    for (const auto &[key, value] : t) {
+      // comma, quotes, colon
+      bound += 4 + 6 * std::string_view(key).size() + size_bound(value);
+    }
+    return bound;
+  } else if constexpr (concepts::container_but_not_string<T> || concepts::appendable_containers<T>) {
+    using value_type = std::remove_cvref_t<std::ranges::range_value_t<T>>;
+    if constexpr (std::is_arithmetic_v<value_type> && !std::is_same_v<value_type, char>) {
+      // A fixed bound per element: no need to visit them.
+      return 2 + size_t(std::ranges::distance(t)) * (1 + size_bound(value_type{}));
+    } else {
+      size_t bound = 2;
+#if defined(__GNUC__) && !defined(__clang__)
+#pragma GCC novector // a vector loop is slower on short containers
+#endif
+#pragma GCC unroll 4
+      for (const auto &item : t) {
+        bound += 1 + size_bound(item);
+      }
+      return bound;
+    }
+  } else if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag)) {
+    constexpr auto dm = simdjson::detail::transparent_member(^^T);
+    return size_bound(t.[:dm:]);
+  } else {
+    return 2 + fields_bound(t);
+  }
+}
+
+} // namespace bound_detail
+
+// Write t through an unchecked writer when its size bound is available,
+// reserving that many bytes first, and through the checked writer otherwise.
+template <class T>
+simdjson_really_inline void append_bounded(string_builder &b, const T &t) {
+  // On 32-bit systems, the bound could overflow: keep the checked writer.
+  if constexpr (sizeof(size_t) >= 8 && bound_detail::is_bounded<T>()) {
+    const size_t bound = bound_detail::size_bound(t) + unchecked_slack;
+    const size_t pos = b.unsafe_position();
+    // The bound is a sum of in-memory sizes times a small constant: it cannot
+    // overflow on a 64-bit system. Be pedantic elsewhere.
+    if (sizeof(size_t) >= 8 || bound <= (std::numeric_limits<size_t>::max)() - pos) {
+      const size_t cap = b.unsafe_capacity();
+      // Grow geometrically so that many small appends stay amortized.
+      if (pos + bound <= cap || b.unsafe_grow((std::max)(cap * 2, pos + bound))) {
+        unchecked_writer w(b.unsafe_data(), pos);
+        atom(w, t);
+        b.unsafe_set_position(w.pos);
+      }
+      return;
+    }
   }
-  b.append(']');
+  writer w(b);
+  atom(w, t);
+  w.sync();
 }

-// append functions that delegate to atom functions for primitive types
+// append() -- top-level entry. Each overload constructs a stack-local
+// writer, runs atom(w, t) through the inlined call chain, then syncs
+// the local position back into the string_builder.
 template <class T>
   requires(std::is_arithmetic_v<T> && !std::is_same_v<T, char>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  writer w(b);
+  atom(w, t);
+  w.sync();
 }

 template <class T>
@@ -55010,20 +67254,22 @@ template <class T>
            std::is_same_v<T, std::string_view> ||
            std::is_same_v<T, const char *> ||
            std::is_same_v<T, char>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  writer w(b);
+  atom(w, t);
+  w.sync();
 }

 template <concepts::optional_type T>
   requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  append_bounded(b, t);
 }

 template <concepts::smart_pointer T>
   requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  append_bounded(b, t);
 }

 template <concepts::appendable_containers T>
@@ -55031,14 +67277,14 @@ template <concepts::appendable_containers T>
            !concepts::optional_type<T> && !concepts::smart_pointer<T> &&
            !std::is_same_v<T, std::string> &&
            !std::is_same_v<T, std::string_view> && !std::is_same_v<T, const char*> && !require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  append_bounded(b, t);
 }

 template <concepts::string_view_keyed_map T>
   requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  append_bounded(b, t);
 }

 // works for struct
@@ -55052,39 +67298,15 @@ template <class Z>
            !std::is_same_v<Z, std::string_view> &&
            !std::is_same_v<Z, const char*> &&
            !std::is_same_v<Z, char> && !require_custom_serialization<Z>)
-void append(string_builder &b, const Z &z) {
-  int i = 0;
-  b.append('{');
-  template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^Z, std::meta::access_context::unchecked()))) {
-    if (i != 0)
-      b.append(',');
-    constexpr auto key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(dm)));
-    b.append_raw(key);
-    b.append(':');
-    atom(b, z.[:dm:]);
-    i++;
-  };
-  b.append('}');
+simdjson_inline void append(string_builder &b, const Z &z) {
+  append_bounded(b, z);
 }

 // works for container that have begin() and end() iterators
 template <class Z>
   requires(concepts::container_but_not_string<Z> && !require_custom_serialization<Z>)
-void append(string_builder &b, const Z &z) {
-  auto it = z.begin();
-  auto end = z.end();
-  if (it == end) {
-    b.append_raw("[]");
-    return;
-  }
-  b.append('[');
-  atom(b, *it);
-  ++it;
-  for (; it != end; ++it) {
-    b.append(',');
-    atom(b, *it);
-  }
-  b.append(']');
+simdjson_inline void append(string_builder &b, const Z &z) {
+  append_bounded(b, z);
 }

 template <class Z>
@@ -55095,22 +67317,40 @@ void append(string_builder &b, const Z &z) {


 template <class Z>
-simdjson_warn_unused simdjson_result<std::string> to_json_string(const Z &z, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
-  string_builder b(initial_capacity);
-  append(b, z);
-  std::string_view s;
-  if(auto e = b.view().get(s); e) { return e; }
-  return std::string(s);
+simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
+  if constexpr (sizeof(size_t) >= 8 && bound_detail::is_bounded<Z>()) {
+    // Write straight into s, sized by the bound: no intermediate buffer, no copy.
+    // Prior related work: jsonifier's serializeJson resizes once through
+    // resize_and_overwrite (serializer.hpp).
+    (void)initial_capacity;
+    const size_t bound = bound_detail::size_bound(z) + unchecked_slack;
+    auto write = [&z](char *p) noexcept {
+      unchecked_writer w(p, 0);
+      atom(w, z);
+      return w.pos;
+    };
+#if defined(__cpp_lib_string_resize_and_overwrite) && __cpp_lib_string_resize_and_overwrite >= 202110L
+    s.resize_and_overwrite(bound, [&write](char *p, size_t) noexcept { return write(p); });
+#else
+    s.resize(bound);
+    s.resize(write(s.data()));
+#endif
+    return SUCCESS;
+  } else {
+    string_builder b(initial_capacity);
+    append(b, z);
+    std::string_view view;
+    if(auto e = b.view().get(view); e) { return e; }
+    s.assign(view);
+    return SUCCESS;
+  }
 }

 template <class Z>
-simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
-  string_builder b(initial_capacity);
-  append(b, z);
-  std::string_view view;
-  if(auto e = b.view().get(view); e) { return e; }
-  s.assign(view);
-  return SUCCESS;
+simdjson_warn_unused simdjson_result<std::string> to_json_string(const Z &z, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
+  std::string s;
+  if(auto e = to_json(z, s, initial_capacity); e) { return e; }
+  return s;
 }

 template <class Z>
@@ -55123,40 +67363,41 @@ string_builder& operator<<(string_builder& b, const Z& z) {
 template<constevalutil::fixed_string... FieldNames, typename T>
   requires(std::is_class_v<T> && (sizeof...(FieldNames) > 0))
 void extract_from(string_builder &b, const T &obj) {
-  // Helper to check if a field name matches any of the requested fields
-  auto should_extract = [](std::string_view field_name) constexpr -> bool {
-    return ((FieldNames.view() == field_name) || ...);
-  };
-
-  b.append('{');
+  writer w(b);
+  if (!w.ensure(1)) { w.sync(); return; }
+  w.ptr[w.pos++] = '{';
   bool first = true;
-
   // Iterate through all members of T using reflection
-  template for (constexpr auto mem : std::define_static_array(
-      std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
-
+  static constexpr auto members = std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()));
+  template for (constexpr auto mem : members) {
     if constexpr (std::meta::is_public(mem)) {
-      constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
+      static constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));

       // Only serialize this field if it's in our list of requested fields
-      if constexpr (should_extract(key)) {
-        if (!first) {
-          b.append(',');
+      if constexpr (((FieldNames.view() == key) || ...)) {
+        static constexpr auto first_key = std::define_static_string(
+            constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)) + ":");
+        static constexpr auto rest_key = std::define_static_string(
+            std::string(",") + constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)) + ":");
+        constexpr size_t first_key_len = std::char_traits<char>::length(first_key);
+        constexpr size_t rest_key_len = std::char_traits<char>::length(rest_key);
+        if (!w.ensure(rest_key_len)) { w.sync(); return; }
+        if (first) {
+          std::memcpy(w.ptr + w.pos, first_key, first_key_len);
+          w.pos += first_key_len;
+        } else {
+          std::memcpy(w.ptr + w.pos, rest_key, rest_key_len);
+          w.pos += rest_key_len;
         }
         first = false;
-
-        // Serialize the key
-        constexpr auto quoted_key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)));
-        b.append_raw(quoted_key);
-        b.append(':');
-
-        // Serialize the value
-        atom(b, obj.[:mem:]);
+        atom(w, obj.[:mem:]);
       }
     }
   };

-  b.append('}');
+  if (!w.ensure(1)) { w.sync(); return; }
+  w.ptr[w.pos++] = '}';
+  w.sync();
 }

 template<constevalutil::fixed_string... FieldNames, typename T>
@@ -55169,25 +67410,18 @@ simdjson_warn_unused simdjson_result<std::string> extract_from(const T &obj, siz
   return std::string(s);
 }

+SIMDJSON_POP_DISABLE_WARNINGS
+
 } // namespace builder
 } // namespace lsx
 // Alias the function template to 'to' in the global namespace
 template <class Z>
 simdjson_warn_unused simdjson_result<std::string> to_json(const Z &z, size_t initial_capacity = lsx::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
-  lsx::builder::string_builder b(initial_capacity);
-  lsx::builder::append(b, z);
-  std::string_view s;
-  if(auto e = b.view().get(s); e) { return e; }
-  return std::string(s);
+  return lsx::builder::to_json_string(z, initial_capacity);
 }
 template <class Z>
 simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = lsx::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
-  lsx::builder::string_builder b(initial_capacity);
-  lsx::builder::append(b, z);
-  std::string_view view;
-  if(auto e = b.view().get(view); e) { return e; }
-  s.assign(view);
-  return SUCCESS;
+  return lsx::builder::to_json(z, s, initial_capacity);
 }
 // Global namespace function for extract_from
 template<constevalutil::fixed_string... FieldNames, typename T>
@@ -55333,6 +67567,7 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
 /* including simdjson/generic/builder/json_string_builder-inl.h for lsx: #include "simdjson/generic/builder/json_string_builder-inl.h" */
 /* begin file simdjson/generic/builder/json_string_builder-inl.h for lsx */
 #include <array>
+#include <cmath>
 #include <cstring>
 #include <limits>
 #include <type_traits>
@@ -55367,6 +67602,11 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
 #define SIMDJSON_EXPERIMENTAL_HAS_LSX 1
 #endif
 #endif
+#if defined(__loongarch_asx)
+#ifndef SIMDJSON_EXPERIMENTAL_HAS_LASX
+#define SIMDJSON_EXPERIMENTAL_HAS_LASX 1
+#endif
+#endif
 #if defined(__riscv_v_intrinsic) && __riscv_v_intrinsic >= 11000 &&            \
     defined(__riscv_vector)
 #ifndef SIMDJSON_EXPERIMENTAL_HAS_RVV
@@ -55386,6 +67626,9 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
 #endif
 #if SIMDJSON_EXPERIMENTAL_HAS_SSE2
 #include <emmintrin.h>
+#if defined(__AVX2__)
+#include <immintrin.h>
+#endif
 #ifdef _MSC_VER
 #include <intrin.h>
 #endif
@@ -55393,6 +67636,9 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
 #if SIMDJSON_EXPERIMENTAL_HAS_LSX
 #include <lsxintrin.h>
 #endif
+#if SIMDJSON_EXPERIMENTAL_HAS_LASX
+#include <lasxintrin.h>
+#endif
 #if SIMDJSON_EXPERIMENTAL_HAS_RVV
 #include <riscv_vector.h>
 #endif
@@ -55442,105 +67688,6 @@ inline bool has_json_escapable_byte(uint64_t x) {

 **/

-SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline bool
-simple_needs_escaping(std::string_view v) {
-  for (char c : v) {
-    // a table lookup is faster than a series of comparisons
-    if (json_quotable_character[static_cast<uint8_t>(c)]) {
-      return true;
-    }
-  }
-  return false;
-}
-
-#if SIMDJSON_EXPERIMENTAL_HAS_NEON
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
-  if (view.size() < 16) {
-    return simple_needs_escaping(view);
-  }
-  size_t i = 0;
-  uint8x16_t running = vdupq_n_u8(0);
-  uint8x16_t v34 = vdupq_n_u8(34);
-  uint8x16_t v92 = vdupq_n_u8(92);
-
-  for (; i + 15 < view.size(); i += 16) {
-    uint8x16_t word = vld1q_u8((const uint8_t *)view.data() + i);
-    running = vorrq_u8(running, vceqq_u8(word, v34));
-    running = vorrq_u8(running, vceqq_u8(word, v92));
-    running = vorrq_u8(running, vcltq_u8(word, vdupq_n_u8(32)));
-  }
-  if (i < view.size()) {
-    uint8x16_t word =
-        vld1q_u8((const uint8_t *)view.data() + view.length() - 16);
-    running = vorrq_u8(running, vceqq_u8(word, v34));
-    running = vorrq_u8(running, vceqq_u8(word, v92));
-    running = vorrq_u8(running, vcltq_u8(word, vdupq_n_u8(32)));
-  }
-  return vmaxvq_u32(vreinterpretq_u32_u8(running)) != 0;
-}
-#elif SIMDJSON_EXPERIMENTAL_HAS_SSE2
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
-  if (view.size() < 16) {
-    return simple_needs_escaping(view);
-  }
-  size_t i = 0;
-  __m128i running = _mm_setzero_si128();
-  for (; i + 15 < view.size(); i += 16) {
-
-    __m128i word =
-        _mm_loadu_si128(reinterpret_cast<const __m128i *>(view.data() + i));
-    running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(34)));
-    running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(92)));
-    running = _mm_or_si128(
-        running, _mm_cmpeq_epi8(_mm_subs_epu8(word, _mm_set1_epi8(31)),
-                                _mm_setzero_si128()));
-  }
-  if (i < view.size()) {
-    __m128i word = _mm_loadu_si128(
-        reinterpret_cast<const __m128i *>(view.data() + view.length() - 16));
-    running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(34)));
-    running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(92)));
-    running = _mm_or_si128(
-        running, _mm_cmpeq_epi8(_mm_subs_epu8(word, _mm_set1_epi8(31)),
-                                _mm_setzero_si128()));
-  }
-  return _mm_movemask_epi8(running) != 0;
-}
-#elif SIMDJSON_EXPERIMENTAL_HAS_PPC64
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
-  if (view.size() < 16) {
-    return simple_needs_escaping(view);
-  }
-  size_t i = 0;
-  __vector unsigned char running = vec_splats((unsigned char)0);
-  __vector unsigned char v34 = vec_splats((unsigned char)34);
-  __vector unsigned char v92 = vec_splats((unsigned char)92);
-  __vector unsigned char v32 = vec_splats((unsigned char)32);
-
-  for (; i + 15 < view.size(); i += 16) {
-    __vector unsigned char word =
-        vec_vsx_ld(0, reinterpret_cast<const unsigned char *>(view.data() + i));
-    running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v34));
-    running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v92));
-    running = vec_or(running,
-        (__vector unsigned char)vec_cmplt(word, v32));
-  }
-  if (i < view.size()) {
-    __vector unsigned char word = vec_vsx_ld(
-        0, reinterpret_cast<const unsigned char *>(view.data() + view.length() - 16));
-    running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v34));
-    running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v92));
-    running = vec_or(running,
-        (__vector unsigned char)vec_cmplt(word, v32));
-  }
-  return !vec_all_eq(running, vec_splats((unsigned char)0));
-}
-#else
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
-  return simple_needs_escaping(view);
-}
-#endif
-
 // Scalar fallback for finding next quotable character
 SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline size_t
 find_next_json_quotable_character_scalar(const std::string_view view,
@@ -55631,6 +67778,51 @@ find_next_json_quotable_character(const std::string_view view,
   size_t current = len - remaining;
   return find_next_json_quotable_character_scalar(view, current);
 }
+#elif SIMDJSON_EXPERIMENTAL_HAS_LASX
+simdjson_inline size_t
+find_next_json_quotable_character(const std::string_view view,
+                                  size_t location) noexcept {
+  const size_t len = view.size();
+  const uint8_t *ptr =
+      reinterpret_cast<const uint8_t *>(view.data()) + location;
+  size_t remaining = len - location;
+
+  // SIMD constants for characters requiring escape
+  __m256i v34 = __lasx_xvreplgr2vr_b(34);  // '"'
+  __m256i v92 = __lasx_xvreplgr2vr_b(92);  // '\\'
+  __m256i v32 = __lasx_xvreplgr2vr_b(32);  // control char threshold
+
+  while (remaining >= 32) {
+    __m256i word = __lasx_xvld(ptr, 0);
+
+    // Check for quotable characters: '"', '\\', or control chars (< 32)
+    __m256i needs_escape = __lasx_xvseq_b(word, v34);
+    needs_escape = __lasx_xvor_v(needs_escape, __lasx_xvseq_b(word, v92));
+    needs_escape = __lasx_xvor_v(needs_escape, __lasx_xvslt_bu(word, v32));
+
+    if (!__lasx_xbz_v(needs_escape)) {
+      // Found a quotable character - locate it via the four 64-bit lanes
+      uint64_t lane0 = __lasx_xvpickve2gr_du(needs_escape, 0);
+      uint64_t lane1 = __lasx_xvpickve2gr_du(needs_escape, 1);
+      uint64_t lane2 = __lasx_xvpickve2gr_du(needs_escape, 2);
+      uint64_t lane3 = __lasx_xvpickve2gr_du(needs_escape, 3);
+      size_t offset = ptr - reinterpret_cast<const uint8_t *>(view.data());
+      if (lane0 != 0) {
+        return offset + trailing_zeroes(lane0) / 8;
+      } else if (lane1 != 0) {
+        return offset + 8 + trailing_zeroes(lane1) / 8;
+      } else if (lane2 != 0) {
+        return offset + 16 + trailing_zeroes(lane2) / 8;
+      } else {
+        return offset + 24 + trailing_zeroes(lane3) / 8;
+      }
+    }
+    ptr += 32;
+    remaining -= 32;
+  }
+  size_t current = len - remaining;
+  return find_next_json_quotable_character_scalar(view, current);
+}
 #elif SIMDJSON_EXPERIMENTAL_HAS_LSX
 simdjson_inline size_t
 find_next_json_quotable_character(const std::string_view view,
@@ -55786,6 +67978,254 @@ SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline void escape_json_char(char c, char *&o
   }
 }

+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2 || SIMDJSON_EXPERIMENTAL_HAS_NEON
+#define SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE 1
+#endif
+
+#if SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2
+
+using escape_vector = __m128i;
+// Mask bits that each input byte contributes to escape_bitmask().
+static constexpr unsigned escape_mask_bits = 1;
+
+simdjson_inline escape_vector escape_load16(const uint8_t *p) noexcept {
+  return _mm_loadu_si128(reinterpret_cast<const __m128i *>(p));
+}
+
+simdjson_inline void escape_store16(char *out, escape_vector v) noexcept {
+  _mm_storeu_si128(reinterpret_cast<__m128i *>(out), v);
+}
+
+// Builds a vector whose bytes 0..7 come from a and bytes 8..15 from b.
+simdjson_inline escape_vector escape_load8x2(const uint8_t *a,
+                                             const uint8_t *b) noexcept {
+  return _mm_unpacklo_epi64(
+      _mm_loadl_epi64(reinterpret_cast<const __m128i *>(a)),
+      _mm_loadl_epi64(reinterpret_cast<const __m128i *>(b)));
+}
+
+// Builds a vector whose bytes 0..3 come from a and bytes 4..7 from b. The
+// upper half is unspecified; callers only look at the low eight bits.
+simdjson_inline escape_vector escape_load4x2(const uint8_t *a,
+                                             const uint8_t *b) noexcept {
+  int32_t a32, b32;
+  memcpy(&a32, a, 4);
+  memcpy(&b32, b, 4);
+  return _mm_unpacklo_epi32(_mm_cvtsi32_si128(a32), _mm_cvtsi32_si128(b32));
+}
+
+// Sets every bit of byte k when byte k of v is a quotable character ('"',
+// '\\' or a control character).
+simdjson_inline escape_vector escape_flags(escape_vector v) noexcept {
+  const __m128i v34 = _mm_set1_epi8(34); // '"'
+  const __m128i v92 = _mm_set1_epi8(92); // '\\'
+  const __m128i v31 = _mm_set1_epi8(31); // for control char detection
+  __m128i needs_escape = _mm_cmpeq_epi8(v, v34);
+  needs_escape = _mm_or_si128(needs_escape, _mm_cmpeq_epi8(v, v92));
+  return _mm_or_si128(
+      needs_escape, _mm_cmpeq_epi8(_mm_subs_epu8(v, v31), _mm_setzero_si128()));
+}
+
+// True when any byte needs escaping. Kept separate from escape_bitmask
+// because some instruction sets can answer it without leaving the vector
+// register file; on SSE2 the compiler folds the two together.
+simdjson_inline bool escape_any(escape_vector flags) noexcept {
+  return _mm_movemask_epi8(flags) != 0;
+}
+
+// A 16-bit mask with bit k set when byte k needed escaping.
+simdjson_inline uint64_t escape_bitmask(escape_vector flags) noexcept {
+  return uint64_t(uint32_t(_mm_movemask_epi8(flags)));
+}
+
+#else // SIMDJSON_EXPERIMENTAL_HAS_NEON
+
+using escape_vector = uint8x16_t;
+static constexpr unsigned escape_mask_bits = 4;
+
+simdjson_inline escape_vector escape_load16(const uint8_t *p) noexcept {
+  return vld1q_u8(p);
+}
+
+simdjson_inline void escape_store16(char *out, escape_vector v) noexcept {
+  vst1q_u8(reinterpret_cast<uint8_t *>(out), v);
+}
+
+simdjson_inline escape_vector escape_load8x2(const uint8_t *a,
+                                             const uint8_t *b) noexcept {
+  uint64_t a64, b64;
+  memcpy(&a64, a, 8);
+  memcpy(&b64, b, 8);
+  return vreinterpretq_u8_u64(vsetq_lane_u64(b64, vdupq_n_u64(a64), 1));
+}
+
+simdjson_inline escape_vector escape_load4x2(const uint8_t *a,
+                                             const uint8_t *b) noexcept {
+  uint32_t a32, b32;
+  memcpy(&a32, a, 4);
+  memcpy(&b32, b, 4);
+  return vreinterpretq_u8_u32(vsetq_lane_u32(b32, vdupq_n_u32(a32), 1));
+}
+
+simdjson_inline escape_vector escape_flags(escape_vector v) noexcept {
+  uint8x16_t needs_escape = vceqq_u8(v, vdupq_n_u8(34));              // '"'
+  needs_escape = vorrq_u8(needs_escape, vceqq_u8(v, vdupq_n_u8(92))); // '\\'
+  return vorrq_u8(needs_escape, vcltq_u8(v, vdupq_n_u8(32)));
+}
+
+simdjson_inline bool escape_any(escape_vector flags) noexcept {
+  return vmaxvq_u32(vreinterpretq_u32_u8(flags)) != 0;
+}
+
+// Four bits per byte rather than one: escape_block and the tail paths scale
+// their shifts by escape_mask_bits to match.
+simdjson_inline uint64_t escape_bitmask(escape_vector flags) noexcept {
+  uint8x8_t narrowed = vshrn_n_u16(vreinterpretq_u16_u8(flags), 4);
+  return vget_lane_u64(vreinterpret_u64_u8(narrowed), 0);
+}
+
+#endif // instruction set selection
+
+// The escape bitmask of a 16-byte block.
+simdjson_inline uint64_t escape_mask(escape_vector v) noexcept {
+  return escape_bitmask(escape_flags(v));
+}
+
+// Copies n bytes with n < 16, using overlapping loads and stores. It never
+// reads more than n bytes from src, nor writes more than n bytes to dst.
+simdjson_inline void copy_lt16(char *dst, const uint8_t *src,
+                               size_t n) noexcept {
+  if (n >= 8) {
+    memcpy(dst, src, 8);
+    memcpy(dst + n - 8, src + n - 8, 8);
+  } else if (n >= 4) {
+    memcpy(dst, src, 4);
+    memcpy(dst + n - 4, src + n - 4, 4);
+  } else if (n > 0) {
+    dst[0] = char(src[0]);
+    dst[n >> 1] = char(src[n >> 1]);
+    dst[n - 1] = char(src[n - 1]);
+  }
+}
+
+// Escapes the bytes of src in the range [i, blockend), given that m is the
+// (non-zero) escape mask for that range: the escape_mask_bits-wide lane of m
+// at byte k is non-zero when src[i + k] requires escaping. Returns the updated
+// output pointer.
+simdjson_never_inline char *escape_block(const uint8_t *src, char *out,
+                                         size_t i, size_t blockend,
+                                         uint64_t m) noexcept {
+  constexpr uint64_t lane = (uint64_t(1) << escape_mask_bits) - 1;
+
+  size_t pos = i; // first byte not yet copied
+  while (m) {
+    const size_t tz = trailing_zeroes(m);
+    const size_t next = i + tz / escape_mask_bits;
+    // Copy the run of safe bytes that precedes this escape.
+    copy_lt16(out, src + pos, next - pos);
+    out += next - pos;
+    escape_json_char(char(src[next]), out);
+    pos = next + 1;
+    m &= ~(lane << tz);
+  }
+  // Copy whatever follows the last escape.
+  copy_lt16(out, src + pos, blockend - pos);
+  return out + (blockend - pos);
+}
+
+// Writes the escaped version of input to out, returning the number of bytes
+// written.
+simdjson_really_inline size_t write_string_escaped(const std::string_view input, char *out) {
+  const size_t len = input.size();
+  const uint8_t *src = reinterpret_cast<const uint8_t *>(input.data());
+  const char *const initout = out;
+
+  size_t i = 0;
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2 && defined(__AVX2__)
+  while (i + 32 <= len) {
+    const __m256i word = _mm256_loadu_si256(reinterpret_cast<const __m256i *>(src + i));
+    const __m256i flags = _mm256_or_si256(
+        _mm256_or_si256(_mm256_cmpeq_epi8(word, _mm256_set1_epi8(34)),   // '"'
+                        _mm256_cmpeq_epi8(word, _mm256_set1_epi8(92))),  // '\\'
+        _mm256_cmpeq_epi8(_mm256_subs_epu8(word, _mm256_set1_epi8(31)),
+                          _mm256_setzero_si256()));                      // control
+    const uint32_t mask = uint32_t(_mm256_movemask_epi8(flags));
+    if (simdjson_likely(mask == 0)) {
+      _mm256_storeu_si256(reinterpret_cast<__m256i *>(out), word);
+      out += 32;
+    } else {
+      for (size_t half = 0; half < 32; half += 16) {
+        const uint64_t m = (mask >> half) & 0xFFFF;
+        if (m == 0) {
+          escape_store16(out, escape_load16(src + i + half));
+          out += 16;
+        } else {
+          out = escape_block(src, out, i + half, i + half + 16, m);
+        }
+      }
+    }
+    i += 32;
+  }
+#endif
+  while (i + 16 <= len) {
+    escape_vector word = escape_load16(src + i);
+    escape_vector flags = escape_flags(word);
+    if (simdjson_likely(!escape_any(flags))) {
+      escape_store16(out, word);
+      out += 16;
+    } else {
+      out = escape_block(src, out, i, i + 16, escape_bitmask(flags));
+    }
+    i += 16;
+  }
+  if (i < len) {
+    const size_t rem = len - i;
+    uint64_t m;
+    if (len >= 16) {
+      // The last 16 bytes of the input are in bounds. Bit k of that block's
+      // mask belongs to input position len - 16 + k, so shift it down to align
+      // bit 0 with position i.
+      m = escape_mask(escape_load16(src + len - 16)) >>
+          (escape_mask_bits * (16 - rem));
+    } else if (len >= 8) {
+      // Here i == 0 and rem == len. Two overlapping 8-byte loads cover
+      // [0, 8) and [len - 8, len), which is the whole input since len < 16.
+      uint64_t mm = escape_mask(escape_load8x2(src, src + len - 8));
+      constexpr uint64_t low8 = (uint64_t(1) << (escape_mask_bits * 8)) - 1;
+      m = (mm & low8) |
+          ((mm >> (escape_mask_bits * 8)) << (escape_mask_bits * (len - 8)));
+    } else if (len >= 4) {
+      // Same idea with two overlapping 4-byte loads.
+      uint64_t mm = escape_mask(escape_load4x2(src, src + len - 4));
+      constexpr uint64_t low4 = (uint64_t(1) << (escape_mask_bits * 4)) - 1;
+      m = (mm & low4) | (((mm >> (escape_mask_bits * 4)) & low4)
+                         << (escape_mask_bits * (len - 4)));
+    } else {
+      // Fewer than 4 bytes: at most three table lookups, no need for SIMD.
+      for (size_t k = 0; k < len; k++) {
+        uint8_t c = src[k];
+        if (json_quotable_character[c]) {
+          escape_json_char(char(c), out);
+        } else {
+          *out++ = char(c);
+        }
+      }
+      return size_t(out - initout);
+    }
+    if (m == 0) {
+      copy_lt16(out, src + i, rem);
+      out += rem;
+    } else {
+      out = escape_block(src, out, i, len, m);
+    }
+  }
+  return size_t(out - initout);
+}
+
+#else // SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
 // Writes the escaped version of input to out, returning the number of bytes
 // written. Uses SIMD position finding to locate quotable characters efficiently.
 inline size_t write_string_escaped(const std::string_view input, char *out) {
@@ -55815,9 +68255,13 @@ inline size_t write_string_escaped(const std::string_view input, char *out) {
     escape_json_char(input[location], out);
     location += 1;
   }
-  return out - initout;
+  return size_t(out - initout);
 }

+#endif // SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+#undef SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
+
 simdjson_inline string_builder::string_builder(size_t initial_capacity)
     : buffer(new(std::nothrow) char[initial_capacity]), position(0),
       capacity(buffer.get() != nullptr ? initial_capacity : 0),
@@ -55840,7 +68284,7 @@ simdjson_inline bool string_builder::capacity_check(size_t upcoming_bytes) {
   return is_valid;
 }

-simdjson_inline void string_builder::grow_buffer(size_t desired_capacity) {
+inline void string_builder::grow_buffer(size_t desired_capacity) {
   if (!is_valid) {
     return;
   }
@@ -55896,81 +68340,136 @@ simdjson_inline void string_builder::clear() noexcept {

 namespace internal {

-template <typename number_type, typename = typename std::enable_if<
-                                    std::is_unsigned<number_type>::value>::type>
-simdjson_really_inline int int_log2(number_type x) {
-  return 63 - leading_zeroes(uint64_t(x) | 1);
-}
-
-simdjson_really_inline int fast_digit_count_32(uint32_t x) {
-  static uint64_t table[] = {
-      4294967296,  8589934582,  8589934582,  8589934582,  12884901788,
-      12884901788, 12884901788, 17179868184, 17179868184, 17179868184,
-      21474826480, 21474826480, 21474826480, 21474826480, 25769703776,
-      25769703776, 25769703776, 30063771072, 30063771072, 30063771072,
-      34349738368, 34349738368, 34349738368, 34349738368, 38554705664,
-      38554705664, 38554705664, 41949672960, 41949672960, 41949672960,
-      42949672960, 42949672960};
-  return uint32_t((x + table[int_log2(x)]) >> 32);
-}
-
-simdjson_really_inline int fast_digit_count_64(uint64_t x) {
-  static uint64_t table[] = {9,
-                             99,
-                             999,
-                             9999,
-                             99999,
-                             999999,
-                             9999999,
-                             99999999,
-                             999999999,
-                             9999999999,
-                             99999999999,
-                             999999999999,
-                             9999999999999,
-                             99999999999999,
-                             999999999999999ULL,
-                             9999999999999999ULL,
-                             99999999999999999ULL,
-                             999999999999999999ULL,
-                             9999999999999999999ULL};
-  int y = (19 * int_log2(x) >> 6);
-  y += x > table[y];
-  return y + 1;
-}
-
-template <typename number_type, typename = typename std::enable_if<
-                                    std::is_unsigned<number_type>::value>::type>
-simdjson_really_inline size_t digit_count(number_type v) noexcept {
-  static_assert(sizeof(number_type) == 8 || sizeof(number_type) == 4 ||
-                    sizeof(number_type) == 2 || sizeof(number_type) == 1,
-                "We only support 8-bit, 16-bit, 32-bit and 64-bit numbers");
-  SIMDJSON_IF_CONSTEXPR(sizeof(number_type) <= 4) {
-    return fast_digit_count_32(static_cast<uint32_t>(v));
+// Integer to decimal: James Edward Anhalt III's algorithm
+static const char jeaiii_dd[201] =
+    "00010203040506070809101112131415161718192021222324252627282930313233343536373839"
+    "40414243444546474849505152535455565758596061626364656667686970717273747576777879"
+    "8081828384858687888990919293949596979899";
+static const char jeaiii_fd[201] =
+    "0\0" "1\0" "2\0" "3\0" "4\0" "5\0" "6\0" "7\0" "8\0" "9\0"
+    "10111213141516171819202122232425262728293031323334353637383940414243444546474849"
+    "50515253545556575859606162636465666768697071727374757677787980818283848586878889"
+    "90919293949596979899";
+
+simdjson_really_inline void jeaiii_write_dd(char *p, uint64_t k) noexcept {
+  std::memcpy(p, &jeaiii_dd[2 * k], 2);
+}
+simdjson_really_inline void jeaiii_write_fd(char *p, uint64_t k) noexcept {
+  std::memcpy(p, &jeaiii_fd[2 * k], 2);
+}
+
+// Caller guarantees n < 10^8. Writes 1 to 8 digits.
+simdjson_really_inline char *jeaiii_lt1e8(char *b, uint32_t n) noexcept {
+  constexpr uint64_t mask24 = (uint64_t(1) << 24) - 1;
+  constexpr uint64_t mask32 = (uint64_t(1) << 32) - 1;
+  if (n < 100) {
+    jeaiii_write_fd(b, n);
+    return n < 10 ? b + 1 : b + 2;
+  }
+  if (n < 1000000) {
+    if (n < 10000) {
+      const uint32_t f0 = uint32_t(10 * (1 << 24) / 1e3 + 1) * n;
+      jeaiii_write_fd(b, f0 >> 24);
+      b -= n < 1000;
+      const uint32_t f2 = uint32_t(f0 & mask24) * 100;
+      jeaiii_write_dd(b + 2, f2 >> 24);
+      return b + 4;
+    }
+    const uint64_t f0 = uint64_t(10 * (1ull << 32) / 1e5 + 1) * n;
+    jeaiii_write_fd(b, f0 >> 32);
+    b -= n < 100000;
+    const uint64_t f2 = (f0 & mask32) * 100;
+    jeaiii_write_dd(b + 2, f2 >> 32);
+    const uint64_t f4 = (f2 & mask32) * 100;
+    jeaiii_write_dd(b + 4, f4 >> 32);
+    return b + 6;
+  }
+  const uint64_t f0 = uint64_t(10 * (1ull << 48) / 1e7 + 1) * n >> 16;
+  jeaiii_write_fd(b, f0 >> 32);
+  b -= n < 10000000;
+  const uint64_t f2 = (f0 & mask32) * 100;
+  jeaiii_write_dd(b + 2, f2 >> 32);
+  const uint64_t f4 = (f2 & mask32) * 100;
+  jeaiii_write_dd(b + 4, f4 >> 32);
+  const uint64_t f6 = (f4 & mask32) * 100;
+  jeaiii_write_dd(b + 6, f6 >> 32);
+  return b + 8;
+}
+
+// Caller guarantees z < 10^8. Always writes exactly 8 digits.
+simdjson_really_inline char *jeaiii_8_digits(char *b, uint32_t z) noexcept {
+  constexpr uint64_t mask32 = (uint64_t(1) << 32) - 1;
+  const uint64_t f0 = (uint64_t((1ull << 48) / 1e6 + 1) * z >> 16) + 1;
+  jeaiii_write_dd(b, f0 >> 32);
+  const uint64_t f2 = (f0 & mask32) * 100;
+  jeaiii_write_dd(b + 2, f2 >> 32);
+  const uint64_t f4 = (f2 & mask32) * 100;
+  jeaiii_write_dd(b + 4, f4 >> 32);
+  const uint64_t f6 = (f4 & mask32) * 100;
+  jeaiii_write_dd(b + 6, f6 >> 32);
+  return b + 8;
+}
+
+// Caller guarantees 10^8 <= n < 2^32. Writes 9 or 10 digits.
+simdjson_really_inline char *jeaiii_9_or_10(char *b, uint64_t n) noexcept {
+  constexpr uint64_t mask57 = (uint64_t(1) << 57) - 1;
+  const uint64_t f0 = uint64_t(10 * (1ull << 57) / 1e9 + 1) * n;
+  jeaiii_write_fd(b, f0 >> 57);
+  b -= n < 1000000000;
+  const uint64_t f2 = (f0 & mask57) * 100;
+  jeaiii_write_dd(b + 2, f2 >> 57);
+  const uint64_t f4 = (f2 & mask57) * 100;
+  jeaiii_write_dd(b + 4, f4 >> 57);
+  const uint64_t f6 = (f4 & mask57) * 100;
+  jeaiii_write_dd(b + 6, f6 >> 57);
+  const uint64_t f8 = (f6 & mask57) * 100;
+  jeaiii_write_dd(b + 8, f8 >> 57);
+  return b + 10;
+}
+
+simdjson_really_inline char *write_uint_jeaiii(char *b, uint64_t n) noexcept {
+  if (n < 100000000) {
+    return jeaiii_lt1e8(b, uint32_t(n));
+  }
+  if (n < (uint64_t(1) << 32)) {
+    return jeaiii_9_or_10(b, n);
+  }
+  // At least 10 digits: the low 8 digits, and 2 to 12 digits above them.
+  const uint32_t z = uint32_t(n % 100000000);
+  uint64_t u = n / 100000000;
+  if (u < 100000000) {
+    // u has 2 to 8 digits (if u < 10, n would be below 2^32).
+    b = jeaiii_lt1e8(b, uint32_t(u));
+  } else if (u < (uint64_t(1) << 32)) {
+    b = jeaiii_9_or_10(b, u);
+  } else {
+    // u has 11 or 12 digits: split off 8 more.
+    const uint32_t y = uint32_t(u % 100000000);
+    u /= 100000000;
+    b = jeaiii_lt1e8(b, uint32_t(u)); // 3 or 4 digits
+    b = jeaiii_8_digits(b, y);
   }
-  else {
-    return fast_digit_count_64(static_cast<uint64_t>(v));
-  }
-}
-static const char decimal_table[200] = {
-    0x30, 0x30, 0x30, 0x31, 0x30, 0x32, 0x30, 0x33, 0x30, 0x34, 0x30, 0x35,
-    0x30, 0x36, 0x30, 0x37, 0x30, 0x38, 0x30, 0x39, 0x31, 0x30, 0x31, 0x31,
-    0x31, 0x32, 0x31, 0x33, 0x31, 0x34, 0x31, 0x35, 0x31, 0x36, 0x31, 0x37,
-    0x31, 0x38, 0x31, 0x39, 0x32, 0x30, 0x32, 0x31, 0x32, 0x32, 0x32, 0x33,
-    0x32, 0x34, 0x32, 0x35, 0x32, 0x36, 0x32, 0x37, 0x32, 0x38, 0x32, 0x39,
-    0x33, 0x30, 0x33, 0x31, 0x33, 0x32, 0x33, 0x33, 0x33, 0x34, 0x33, 0x35,
-    0x33, 0x36, 0x33, 0x37, 0x33, 0x38, 0x33, 0x39, 0x34, 0x30, 0x34, 0x31,
-    0x34, 0x32, 0x34, 0x33, 0x34, 0x34, 0x34, 0x35, 0x34, 0x36, 0x34, 0x37,
-    0x34, 0x38, 0x34, 0x39, 0x35, 0x30, 0x35, 0x31, 0x35, 0x32, 0x35, 0x33,
-    0x35, 0x34, 0x35, 0x35, 0x35, 0x36, 0x35, 0x37, 0x35, 0x38, 0x35, 0x39,
-    0x36, 0x30, 0x36, 0x31, 0x36, 0x32, 0x36, 0x33, 0x36, 0x34, 0x36, 0x35,
-    0x36, 0x36, 0x36, 0x37, 0x36, 0x38, 0x36, 0x39, 0x37, 0x30, 0x37, 0x31,
-    0x37, 0x32, 0x37, 0x33, 0x37, 0x34, 0x37, 0x35, 0x37, 0x36, 0x37, 0x37,
-    0x37, 0x38, 0x37, 0x39, 0x38, 0x30, 0x38, 0x31, 0x38, 0x32, 0x38, 0x33,
-    0x38, 0x34, 0x38, 0x35, 0x38, 0x36, 0x38, 0x37, 0x38, 0x38, 0x38, 0x39,
-    0x39, 0x30, 0x39, 0x31, 0x39, 0x32, 0x39, 0x33, 0x39, 0x34, 0x39, 0x35,
-    0x39, 0x36, 0x39, 0x37, 0x39, 0x38, 0x39, 0x39,
-};
+  return jeaiii_8_digits(b, z);
+}
+
+// Writes v at p, which must have to_chars_buffer_size bytes available, and
+// returns the end of what was written.
+simdjson_inline char *write_double(char *p, double v) noexcept {
+#if SIMDJSON_ENABLE_NAN_INF
+  if (simdjson_unlikely(!std::isfinite(v))) {
+    if (std::isnan(v)) {
+      std::memcpy(p, "NaN", 3);
+      return p + 3;
+    }
+    if (v < 0) {
+      *p++ = '-';
+    }
+    std::memcpy(p, "Infinity", 8);
+    return p + 8;
+  }
+#endif
+  return simdjson::internal::to_chars(p, nullptr, v);
+}
 } // namespace internal

 template <typename number_type, typename>
@@ -55998,87 +68497,62 @@ simdjson_inline void string_builder::append(number_type v) noexcept {
     }
   }
   else SIMDJSON_IF_CONSTEXPR(std::is_unsigned<number_type>::value) {
-    // Process 4 digits at a time instead of 2, reducing store operations
-    // and divisions by approximately half for large numbers.
     constexpr size_t max_number_size = 20;
     if (capacity_check(max_number_size)) {
       using unsigned_type = typename std::make_unsigned<number_type>::type;
-      unsigned_type pv = static_cast<unsigned_type>(v);
-      size_t dc = internal::digit_count(pv);
-      char *write_pointer = buffer.get() + position + dc - 1;
-
-      // Process 4 digits per iteration for large numbers
-      while (pv >= 10000) {
-        unsigned_type q = pv / 10000;
-        unsigned_type r = pv % 10000;
-        unsigned_type r_hi = r / 100;  // High 2 digits of remainder
-        unsigned_type r_lo = r % 100;  // Low 2 digits of remainder
-        // Write low 2 digits first (rightmost), then high 2 digits
-        memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
-        memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
-        write_pointer -= 4;
-        pv = q;
-      }
-
-      // Handle remaining 1-4 digits with original 2-digit loop
-      while (pv >= 100) {
-        memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
-        write_pointer -= 2;
-        pv /= 100;
-      }
-      if (pv >= 10) {
-        *write_pointer-- = char('0' + (pv % 10));
-        pv /= 10;
-      }
-      *write_pointer = char('0' + pv);
-      position += dc;
+      char* end = internal::write_uint_jeaiii(
+          buffer.get() + position,
+          static_cast<uint64_t>(static_cast<unsigned_type>(v)));
+      position = end - buffer.get();
     }
   }
   else SIMDJSON_IF_CONSTEXPR(std::is_integral<number_type>::value) {
-    // Same 4-digit batching as unsigned path for signed integers
+    // 19 digits (max abs value of int64_t) + optional minus sign.
     constexpr size_t max_number_size = 20;
     if (capacity_check(max_number_size)) {
       using unsigned_type = typename std::make_unsigned<number_type>::type;
       bool negative = v < 0;
-      unsigned_type pv = static_cast<unsigned_type>(v);
-      if (negative) {
-        pv = 0 - pv; // the 0 is for Microsoft
-      }
-      size_t dc = internal::digit_count(pv);
-      // by always writing the minus sign, we avoid the branch.
+      // 0 - pv (rather than -pv) avoids an MSVC unary-minus warning.
+      unsigned_type pv = negative
+          ? unsigned_type(0) - static_cast<unsigned_type>(v)
+          : static_cast<unsigned_type>(v);
+      // Branchless: always write '-', advance only if negative.
       buffer.get()[position] = '-';
-      position += negative ? 1 : 0;
-      char *write_pointer = buffer.get() + position + dc - 1;
-
-      // Process 4 digits per iteration for large numbers
-      while (pv >= 10000) {
-        unsigned_type q = pv / 10000;
-        unsigned_type r = pv % 10000;
-        unsigned_type r_hi = r / 100;
-        unsigned_type r_lo = r % 100;
-        memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
-        memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
-        write_pointer -= 4;
-        pv = q;
-      }
-
-      // Handle remaining 1-4 digits
-      while (pv >= 100) {
-        memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
-        write_pointer -= 2;
-        pv /= 100;
-      }
-      if (pv >= 10) {
-        *write_pointer-- = char('0' + (pv % 10));
-        pv /= 10;
-      }
-      *write_pointer = char('0' + pv);
-      position += dc;
+      position += negative;
+      char* end = internal::write_uint_jeaiii(
+          buffer.get() + position, static_cast<uint64_t>(pv));
+      position = end - buffer.get();
     }
   }
   else SIMDJSON_IF_CONSTEXPR(std::is_floating_point<number_type>::value) {
-    constexpr size_t max_number_size = 24;
+    // Must reserve to_chars_buffer_size (40): only ~24 chars are emitted,
+    // but to_chars over-writes with fixed-size 16/17-byte copies so the
+    // compiler can inline mem* (see simdjson::internal::to_chars_buffer_size).
+    constexpr size_t max_number_size = simdjson::internal::to_chars_buffer_size;
     if (capacity_check(max_number_size)) {
+#if SIMDJSON_ENABLE_NAN_INF
+      // Check if the input might be NaN or infinity
+      if (simdjson_unlikely(!std::isfinite(v))) {
+        if (std::isnan(v)) {
+          constexpr char nan_literal[] = "NaN";
+          constexpr size_t nan_len = sizeof(nan_literal) - 1;
+
+          std::memcpy(buffer.get() + position, nan_literal, nan_len);
+          position += nan_len;
+        } else {
+          constexpr char inf_literal[] = "Infinity";
+          constexpr size_t inf_len = sizeof(inf_literal) - 1;
+          if (v < 0) {
+            buffer.get()[position] = '-';
+            ++position;
+          }
+          std::memcpy(buffer.get() + position, inf_literal, inf_len);
+          position += inf_len;
+        }
+        return;
+      }
+#endif
+
       // We could specialize for float.
       char *end = simdjson::internal::to_chars(buffer.get() + position, nullptr,
                                                double(v));
@@ -56139,7 +68613,9 @@ simdjson_inline void string_builder::escape_and_append_with_quotes() noexcept {
 #endif

 simdjson_inline void string_builder::append_raw(const char *c) noexcept {
-  size_t len = std::strlen(c);
+  // char_traits::length is constexpr; lets the compiler fold the length
+  // when called with a pointer to a compile-time-constant string.
+  size_t len = std::char_traits<char>::length(c);
   append_raw(c, len);
 }

@@ -56158,6 +68634,14 @@ simdjson_inline void string_builder::append_raw(const char *str,
     position += len;
   }
 }
+
+template <size_t N>
+simdjson_inline void string_builder::append_raw_n(const char *str) noexcept {
+  if (capacity_check(N)) {
+    std::memcpy(buffer.get() + position, str, N);
+    position += N;
+  }
+}
 #if SIMDJSON_SUPPORTS_CONCEPTS
 // Support for optional types (std::optional, etc.)
 template <concepts::optional_type T>
@@ -56187,7 +68671,7 @@ simdjson_inline void string_builder::append(const T &value) {
 #if SIMDJSON_SUPPORTS_RANGES && SIMDJSON_SUPPORTS_CONCEPTS
 // Support for range-based appending (std::ranges::view, etc.)
 template <std::ranges::range R>
-  requires(!std::is_convertible<R, std::string_view>::value && !require_custom_serialization<R>)
+  requires(!std::is_convertible<R, std::string_view>::value && !concepts::optional_type<R> && !require_custom_serialization<R>)
 simdjson_inline void string_builder::append(const R &range) noexcept {
   auto it = std::ranges::begin(range);
   auto end = std::ranges::end(range);
@@ -56473,10 +68957,6 @@ simdjson_inline int count_ones(uint64_t input_num) {
   return __lasx_xvpickve2gr_w(__lasx_xvpcnt_d(__m256i(v4u64{input_num, 0, 0, 0})), 0);
 }

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2, uint64_t *result) {
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-}

 } // unnamed namespace
 } // namespace lasx
@@ -57209,7 +69689,7 @@ public:
 #if SIMDJSON_SUPPORTS_RANGES && SIMDJSON_SUPPORTS_CONCEPTS
   // Support for range-based appending (std::ranges::view, etc.)
   template <std::ranges::range R>
-requires (!std::is_convertible<R, std::string_view>::value && !require_custom_serialization<R>)
+requires (!std::is_convertible<R, std::string_view>::value && !concepts::optional_type<R> && !require_custom_serialization<R>)
   simdjson_inline void append(const R &range) noexcept;
 #endif
   /**
@@ -57223,6 +69703,15 @@ requires (!std::is_convertible<R, std::string_view>::value && !require_custom_se
    * There is no UTF-8 validation.
    */
   simdjson_inline void append_raw(const char *str, size_t len) noexcept;
+
+  /**
+   * Append exactly N characters from str. The length is a template parameter
+   * so the compiler can fully inline the memcpy with a compile-time-constant
+   * size, avoiding the libc call. Used for compile-time-constant keys in the
+   * reflection struct atom.
+   */
+  template <size_t N>
+  simdjson_inline void append_raw_n(const char *str) noexcept;
 #if SIMDJSON_EXCEPTIONS
   /**
    * Creates an std::string from the written JSON buffer.
@@ -57274,6 +69763,26 @@ requires (!std::is_convertible<R, std::string_view>::value && !require_custom_se
    */
   simdjson_inline size_t size() const noexcept;

+  // ============================================================
+  // Internal hooks for the position-as-local writer in json_builder.h.
+  // These exist so the reflection atom code can hold buffer pointer,
+  // position and capacity in registers across long write chains rather
+  // than reloading them after every char* write (strict aliasing
+  // forces those reloads when accessed via members of *this). User
+  // code should NOT call these directly.
+  // ============================================================
+  simdjson_inline char *unsafe_data() noexcept { return buffer.get(); }
+  simdjson_inline size_t unsafe_position() const noexcept { return position; }
+  simdjson_inline size_t unsafe_capacity() const noexcept { return capacity; }
+  simdjson_inline void unsafe_set_position(size_t p) noexcept { position = p; }
+  /// Make capacity available for at least `n` more bytes after the current
+  /// position. Returns false if the allocation failed.
+  simdjson_inline bool unsafe_grow(size_t needed_total_capacity) noexcept {
+    grow_buffer(needed_total_capacity);
+    return is_valid;
+  }
+  simdjson_inline bool unsafe_is_valid() const noexcept { return is_valid; }
+
 private:
   /**
    * Returns true if we can write at least upcoming_bytes bytes.
@@ -57287,7 +69796,7 @@ private:
    * If the allocation fails, is_valid is set to false. We expect
    * that this function would not be repeatedly called.
    */
-  simdjson_inline void grow_buffer(size_t desired_capacity);
+  inline void grow_buffer(size_t desired_capacity);

   /**
    * We use this helper function to make sure that is_valid is kept consistent.
@@ -57344,6 +69853,7 @@ simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initi
 /* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_STRING_BUILDER_H */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/builder/json_string_builder.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/concepts.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/annotations.h" */
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
 #if SIMDJSON_STATIC_REFLECTION

@@ -57361,64 +69871,370 @@ namespace simdjson {
 namespace lasx {
 namespace builder {

-template <class T>
+// Forward-declare helpers defined in json_string_builder-inl.h so the
+// writer-based atom code below can call them (the -inl.h is not yet
+// included at the point this header is parsed; without these forwards,
+// name lookup falls back to the wrong outer namespace).
+namespace internal {
+simdjson_really_inline char *write_uint_jeaiii(char *p, uint64_t v) noexcept;
+simdjson_inline char *write_double(char *p, double v) noexcept;
+} // namespace internal
+simdjson_really_inline size_t write_string_escaped(const std::string_view input, char *out);
+
+SIMDJSON_PUSH_DISABLE_WARNINGS
+SIMDJSON_DISABLE_GCC_WARNING(-Warray-bounds)
+#if !defined(__clang__)
+SIMDJSON_DISABLE_GCC_WARNING(-Wstringop-overflow)
+#endif
+
+// =============================================================
+// `writer`: position-as-local hot-path writer used by the reflection
+// atom code below. Holds the buffer pointer, write position and
+// capacity in three fields that, once `writer` itself is a stack-local
+// in the caller and all atom() functions are inlined, become true
+// register-resident locals after SROA. Glaze achieves the same effect
+// by passing `B&& b, auto&& ix` through every helper. Holding `pos`
+// in a register (rather than as a member of string_builder) is what
+// breaks the strict-aliasing penalty on every char* write through the
+// buffer, which forces a reload of `b.position` and `b.capacity`
+// after every byte.
+//
+// basic_writer<false> (below) is the unchecked variant: the caller has
+// already reserved enough capacity for everything the write chain can
+// produce (see bound_detail::size_bound), so ensure() compiles away.
+// =============================================================
+template <bool Checked>
+struct basic_writer {
+  static constexpr bool checked = Checked;
+  char *ptr;        // buffer pointer (refreshed after a grow)
+  size_t pos;       // write position (local)
+  size_t cap;       // capacity (refreshed after a grow)
+  string_builder &sb;  // back-ref for grow / sync
+
+  // Snapshot string_builder state into a writer for the duration of
+  // a write chain.
+  simdjson_really_inline basic_writer(string_builder &builder) noexcept
+      : ptr(builder.unsafe_data())
+      , pos(builder.unsafe_position())
+      , cap(builder.unsafe_capacity())
+      , sb(builder) {}
+
+  // Write the local position back to the underlying string_builder.
+  // Caller is responsible for invoking before the writer is dropped
+  // (otherwise data is lost). Idempotent.
+  simdjson_really_inline void sync() noexcept {
+    sb.unsafe_set_position(pos);
+  }
+
+  // Ensure at least `n` more bytes of free capacity. Grows the
+  // underlying buffer if needed (rare path). Returns false on
+  // allocation failure.
+  simdjson_really_inline bool ensure(size_t n) noexcept {
+    // pos <= cap, and cap is the size of a live allocation, so pos + n
+    // cannot wrap when n is a small constant or a compile-time length.
+    // Callers passing a size derived from input (the string atoms) must
+    // bound it against pos themselves. Keep the `pos + n <= cap` form:
+    // `n <= cap - pos` is measurably slower once the serializer is inlined.
+    if (simdjson_likely(pos + n <= cap)) { return true; }
+    return grow_slow(n);
+  }
+
+  simdjson_never_inline bool grow_slow(size_t n) noexcept {
+    // Detect overflow.
+    // This is pedantic except maybe on 32-bit targets.
+    if (simdjson_unlikely(pos + n < pos)) return false;
+    sb.unsafe_set_position(pos);
+    // even if 2*capacity overflows, the (std::max) below will pick the needed value,
+    // so we do not need a separate overflow check here.
+    if (!sb.unsafe_grow((std::max)(cap * 2, pos + n))) {
+      // The string_builder freed its buffer and is now invalid (null buffer,
+      // zero capacity and position). Mirror that state so that every later
+      // ensure() fails too: callers only return from the current atom, and
+      // their callers keep writing.
+      ptr = nullptr;
+      pos = 0;
+      cap = 0;
+      return false;
+    }
+    ptr = sb.unsafe_data();
+    cap = sb.unsafe_capacity();
+    return true;
+  }
+};
+
+// The unchecked writer writes into a raw buffer that the caller sized with
+// serialized_size_bound: it never grows and needs no string_builder.
+template <>
+struct basic_writer<false> {
+  static constexpr bool checked = false;
+  char *ptr;
+  size_t pos;
+
+  simdjson_really_inline basic_writer(char *buffer, size_t position) noexcept
+      : ptr(buffer), pos(position) {}
+
+  simdjson_really_inline bool ensure(size_t) const noexcept { return true; }
+};
+
+using writer = basic_writer<true>;
+using unchecked_writer = basic_writer<false>;
+
+// Bytes reserved past the size bound for an unchecked writer: it may then
+// write a little past the end of what it produces (e.g., copy keys as whole
+// 16-byte blocks).
+inline constexpr size_t unchecked_slack = 64;
+
+consteval size_t padded_key_length(size_t length) {
+  return (length + 15) / 16 * 16;
+}
+
+// === Helper: invoke a string_builder member that writes variable-length
+// content (escape_and_append_with_quotes etc), syncing the writer's local
+// state before the call and reloading after. Used for string fields where
+// rewriting the entire SIMD escape path through the writer would be a much
+// bigger refactor. f may be user code (a with<Adapter> serializer) that
+// throws: the exception then propagates to the caller.
+template <class W, class F>
+simdjson_really_inline void call_through_string_builder(W &w, F &&f) noexcept(noexcept(f(w.sb))) {
+  w.sync();
+  f(w.sb);
+  w.ptr = w.sb.unsafe_data();
+  w.pos = w.sb.unsafe_position();
+  w.cap = w.sb.unsafe_capacity();
+}
+
+// Helpers implementing the serialization side of the annotations (see
+// simdjson/annotations.h) for reflected structures.
+namespace annotation_detail {
+
+// A member is serialized unless it is annotated with skip or skip_serializing.
+consteval bool is_serialized_member(std::meta::info dm) {
+  return !simdjson::detail::has_annotation(dm, ^^simdjson::detail::skip_tag)
+      && !simdjson::detail::has_annotation(dm, ^^simdjson::detail::skip_serializing_tag);
+}
+
+// False when the member has a skip_serializing_if<pred> annotation and
+// pred(value) is true.
+template <auto dm, typename V>
+simdjson_really_inline bool should_serialize(const V &value) {
+  constexpr std::meta::info skip_if_type = simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::skip_serializing_if_t);
+  if constexpr (skip_if_type != std::meta::info{}) {
+    using skip_if = typename [: skip_if_type :];
+    return !skip_if::predicate(value);
+  } else {
+    (void)value;
+    return true;
+  }
+}
+
+// Serialize a member value, through its with<Adapter> annotation when the
+// adapter provides a serialize function.
+template <auto dm, class W, typename V>
+simdjson_really_inline void atom_member(W &w, const V &value) {
+  constexpr std::meta::info with_type = simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::with_t);
+  if constexpr (with_type != std::meta::info{}) {
+    using adapter = typename [: with_type :]::adapter;
+    if constexpr (requires(string_builder &b) { adapter::serialize(b, value); }) {
+      call_through_string_builder(w, [&](string_builder &b) { adapter::serialize(b, value); });
+    } else {
+      atom(w, value);
+    }
+  } else {
+    atom(w, value);
+  }
+}
+
+// Write the "key":value pairs of the members of t (without the braces), each
+// preceded by a comma unless it is the first one. The members of a member
+// annotated with flatten are written in its place.
+template <class W, class T>
+simdjson_really_inline void atom_fields(W &w, const T &t, bool &first) {
+  // Per-field block: ensure key+value worst case, then write key + value
+  // through the writer's local pos. For arithmetic fields, the integer
+  // write happens directly via write_uint_jeaiii on w.ptr+w.pos, so pos
+  // never round-trips through memory.
+  template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+    if constexpr (is_serialized_member(dm)) {
+      if (should_serialize<dm>(t.[:dm:])) {
+        if constexpr (simdjson::detail::has_annotation(dm, ^^simdjson::detail::flatten_tag)) {
+          static_assert(std::meta::is_class_type(simdjson::detail::flattened_type(dm)));
+          using flattened = std::remove_cvref_t<decltype(t.[:dm:])>;
+          static_assert(!concepts::container_but_not_string<flattened> && !concepts::string_view_keyed_map<flattened> &&
+                        !concepts::appendable_containers<flattened> && !concepts::optional_type<flattened> &&
+                        !concepts::smart_pointer<flattened> && !std::is_same_v<flattened, std::string> &&
+                        !std::is_same_v<flattened, std::string_view> && !require_custom_serialization<flattened>,
+                        "simdjson::flatten requires a member whose type is a structure serialized member by member");
+          atom_fields(w, t.[:dm:], first);
+        } else {
+          // Copy the key as whole 16-byte blocks from a zero-padded copy (one
+          // load and one store); ensure() reserves the padded length, and the
+          // unchecked writer has slack past its bound. Prior related work:
+          // jsonifier copies a power-of-two padded key and advances the cursor
+          // by the real length (serialize_impl.hpp, packed_blitter,
+          // https://github.com/nihilai-collective/Jsonifier).
+          constexpr const char* key_name = simdjson::get_json_key_name<dm>();
+          constexpr size_t first_key_len = constevalutil::consteval_to_quoted_escaped(key_name).size() + 1;
+          constexpr size_t rest_key_len = first_key_len + 1;
+          constexpr auto first_key = std::define_static_string(
+              constevalutil::consteval_to_quoted_escaped(key_name) + ":" +
+              std::string(padded_key_length(first_key_len) - first_key_len, '\0'));
+          constexpr auto rest_key = std::define_static_string(
+              std::string(",") + constevalutil::consteval_to_quoted_escaped(key_name) + ":" +
+              std::string(padded_key_length(rest_key_len) - rest_key_len, '\0'));
+          if (!w.ensure(padded_key_length(rest_key_len))) { return; }
+          if (first) {
+            std::memcpy(w.ptr + w.pos, first_key, padded_key_length(first_key_len));
+            w.pos += first_key_len;
+          } else {
+            std::memcpy(w.ptr + w.pos, rest_key, padded_key_length(rest_key_len));
+            w.pos += rest_key_len;
+          }
+          first = false;
+          atom_member<dm>(w, t.[:dm:]);
+        }
+      }
+    }
+  };
+}
+
+} // namespace annotation_detail
+
+template <class W, class T>
   requires(concepts::container_but_not_string<T> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &t) {
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
   auto it = t.begin();
   auto end = t.end();
   if (it == end) {
-    b.append_raw("[]");
+    if (!w.ensure(2)) return;
+    std::memcpy(w.ptr + w.pos, "[]", 2);
+    w.pos += 2;
     return;
   }
-  b.append('[');
-  atom(b, *it);
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = '[';
+  atom(w, *it);
   ++it;
   for (; it != end; ++it) {
-    b.append(',');
-    atom(b, *it);
+    if (!w.ensure(1)) return;
+    w.ptr[w.pos++] = ',';
+    atom(w, *it);
   }
-  b.append(']');
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = ']';
 }

-template <class T>
+template <class W, class T>
   requires(std::is_same_v<T, std::string> ||
            std::is_same_v<T, std::string_view> ||
            std::is_same_v<T, const char *> ||
            std::is_same_v<T, char>)
-constexpr void atom(string_builder &b, const T &t) {
-  b.escape_and_append_with_quotes(t);
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
+  // Inline the escape path through the writer so we never round-trip
+  // pos through memory for string fields (Twitter is dominated by
+  // these -- sync/reload around each string was a real cost).
+  std::string_view input;
+  if constexpr (std::is_same_v<T, char>) {
+    input = std::string_view(&t, 1);
+  } else {
+    input = std::string_view(t);
+  }
+  // Worst-case escape: every byte expands to \uXXXX (6 chars), plus 2 quotes.
+  // Guard against w.pos + 2 + 6 * input.size() wrapping for huge inputs -- if
+  // it wrapped to a small value, ensure() would spuriously succeed and the
+  // subsequent escape would overflow the buffer. max - w.pos cannot wrap, and
+  // size < (max - pos) / 6 implies pos + 6 * size + 6 <= max.
+  // Note that this is pedantic except maybe on 32-bit targets.
+  if constexpr (W::checked) {
+    if (simdjson_unlikely(input.size() >= ((std::numeric_limits<size_t>::max)() - w.pos) / 6)) { return; }
+    if (!w.ensure(2 + 6 * input.size())) { return; }
+  }
+  w.ptr[w.pos++] = '"';
+  w.pos += write_string_escaped(input, w.ptr + w.pos);
+  w.ptr[w.pos++] = '"';
 }

-template <concepts::string_view_keyed_map T>
+template <class W, concepts::string_view_keyed_map T>
   requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &m) {
+simdjson_really_inline constexpr void atom(W &w, const T &m) {
   if (m.empty()) {
-    b.append_raw("{}");
+    if (!w.ensure(2)) return;
+    std::memcpy(w.ptr + w.pos, "{}", 2);
+    w.pos += 2;
     return;
   }
-  b.append('{');
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = '{';
   bool first = true;
   for (const auto& [key, value] : m) {
     if (!first) {
-      b.append(',');
+      if (!w.ensure(1)) return;
+      w.ptr[w.pos++] = ',';
     }
     first = false;
-    // Keys must be convertible to string_view per the concept
-    b.escape_and_append_with_quotes(key);
-    b.append(':');
-    atom(b, value);
+    // Keys must be convertible to string_view per the concept.
+    std::string_view key_sv(key);
+    // Guard against w.pos + 3 + 6 * key_sv.size() wrapping for huge keys, if
+    // it wrapped to a small value, ensure() would spuriously succeed and the
+    // subsequent escape would overflow the buffer. max - w.pos cannot wrap, and
+    // size < (max - pos) / 6 implies pos + 6 * size + 6 <= max.
+    // Note that this is pedantic except maybe on 32-bit targets.
+    if constexpr (W::checked) {
+      if (simdjson_unlikely(key_sv.size() >= ((std::numeric_limits<size_t>::max)() - w.pos) / 6)) { return; }
+      if (!w.ensure(2 + 6 * key_sv.size() + 1)) { return; }
+    }
+    w.ptr[w.pos++] = '"';
+    w.pos += write_string_escaped(key_sv, w.ptr + w.pos);
+    w.ptr[w.pos++] = '"';
+    w.ptr[w.pos++] = ':';
+    atom(w, value);
   }
-  b.append('}');
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = '}';
 }


-template<typename number_type,
+template<class W, typename number_type,
          typename = typename std::enable_if<std::is_arithmetic<number_type>::value && !std::is_same_v<number_type, char>>::type>
-constexpr void atom(string_builder &b, const number_type t) {
-  b.append(t);
+simdjson_really_inline constexpr void atom(W &w, const number_type t) {
+  // Booleans / floats: defer to string_builder (rare path; keeps writer hot
+  // path free of float-formatter machinery). For integers, write directly
+  // via jeaiii using local pos.
+  if constexpr (std::is_same_v<number_type, bool>) {
+    if (t) {
+      if (!w.ensure(4)) return;
+      std::memcpy(w.ptr + w.pos, "true", 4);
+      w.pos += 4;
+    } else {
+      if (!w.ensure(5)) return;
+      std::memcpy(w.ptr + w.pos, "false", 5);
+      w.pos += 5;
+    }
+  } else if constexpr (std::is_floating_point_v<number_type>) {
+    if constexpr (W::checked) {
+      call_through_string_builder(w, [&](string_builder &b) { b.append(t); });
+    } else {
+      w.pos = size_t(internal::write_double(w.ptr + w.pos, double(t)) - w.ptr);
+    }
+  } else if constexpr (std::is_unsigned_v<number_type>) {
+    if (!w.ensure(20)) return;
+    char *end = internal::write_uint_jeaiii(
+        w.ptr + w.pos, static_cast<uint64_t>(t));
+    w.pos = static_cast<size_t>(end - w.ptr);
+  } else {
+    // signed integral
+    if (!w.ensure(20)) return;
+    using U = typename std::make_unsigned<number_type>::type;
+    bool negative = t < 0;
+    U pv = negative ? U(0) - static_cast<U>(t) : static_cast<U>(t);
+    w.ptr[w.pos] = '-';
+    w.pos += negative;
+    char *end = internal::write_uint_jeaiii(
+        w.ptr + w.pos, static_cast<uint64_t>(pv));
+    w.pos = static_cast<size_t>(end - w.ptr);
+  }
 }

-template <class T>
+template <class W, class T>
   requires(std::is_class_v<T> && !concepts::container_but_not_string<T> &&
            !concepts::string_view_keyed_map<T> &&
            !concepts::optional_type<T> &&
@@ -57428,92 +70244,259 @@ template <class T>
            !std::is_same_v<T, std::string_view> &&
            !std::is_same_v<T, const char*> &&
            !std::is_same_v<T, char> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &t) {
-  int i = 0;
-  b.append('{');
-  template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
-    if (i != 0)
-      b.append(',');
-    constexpr auto key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(dm)));
-    b.append_raw(key);
-    b.append(':');
-    atom(b, t.[:dm:]);
-    i++;
-  };
-  b.append('}');
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
+  if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag)) {
+    // A transparent structure is serialized as its single member.
+    constexpr auto dm = simdjson::detail::transparent_member(^^T);
+    annotation_detail::atom_member<dm>(w, t.[:dm:]);
+  } else {
+    bool first = true;
+    if (!w.ensure(1)) { return; }
+    w.ptr[w.pos++] = '{';
+    annotation_detail::atom_fields(w, t, first);
+    if (!w.ensure(1)) { return; }
+    w.ptr[w.pos++] = '}';
+  }
 }

 // Support for optional types (std::optional, etc.)
-template <concepts::optional_type T>
+template <class W, concepts::optional_type T>
   requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &opt) {
+simdjson_really_inline constexpr void atom(W &w, const T &opt) {
   if (opt) {
-    atom(b, opt.value());
+    atom(w, opt.value());
   } else {
-    b.append_raw("null");
+    if (!w.ensure(4)) return;
+    std::memcpy(w.ptr + w.pos, "null", 4);
+    w.pos += 4;
   }
 }

 // Support for smart pointers (std::unique_ptr, std::shared_ptr, etc.)
-template <concepts::smart_pointer T>
+template <class W, concepts::smart_pointer T>
   requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &ptr) {
+simdjson_really_inline constexpr void atom(W &w, const T &ptr) {
   if (ptr) {
-    atom(b, *ptr);
+    atom(w, *ptr);
   } else {
-    b.append_raw("null");
+    if (!w.ensure(4)) return;
+    std::memcpy(w.ptr + w.pos, "null", 4);
+    w.pos += 4;
   }
 }

 // Support for enums - serialize as string representation using expand approach from P2996R12
-template <typename T>
+template <class W, typename T>
   requires(std::is_enum_v<T> && !require_custom_serialization<T>)
-void atom(string_builder &b, const T &e) {
+simdjson_really_inline void atom(W &w, const T &e) {
 #if SIMDJSON_STATIC_REFLECTION
-  constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
+  static constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
   template for (constexpr auto enum_val : enumerators) {
-    constexpr auto enum_str = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(enum_val)));
+    constexpr auto enum_str = std::define_static_string(constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<enum_val>()));
+    constexpr size_t enum_str_len = std::char_traits<char>::length(enum_str);
     if (e == [:enum_val:]) {
-      b.append_raw(enum_str);
+      if (!w.ensure(enum_str_len)) return;
+      std::memcpy(w.ptr + w.pos, enum_str, enum_str_len);
+      w.pos += enum_str_len;
       return;
     }
   };
   // Fallback to integer if enum value not found
-  atom(b, static_cast<std::underlying_type_t<T>>(e));
+  atom(w, static_cast<std::underlying_type_t<T>>(e));
 #else
   // Fallback: serialize as integer if reflection not available
-  atom(b, static_cast<std::underlying_type_t<T>>(e));
+  atom(w, static_cast<std::underlying_type_t<T>>(e));
 #endif
 }

 // Support for appendable containers that don't have operator[] (sets, etc.)
-template <concepts::appendable_containers T>
+template <class W, concepts::appendable_containers T>
   requires(!concepts::container_but_not_string<T> && !concepts::string_view_keyed_map<T> &&
            !concepts::optional_type<T> && !concepts::smart_pointer<T> &&
            !std::is_same_v<T, std::string> &&
            !std::is_same_v<T, std::string_view> && !std::is_same_v<T, const char*> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &container) {
+simdjson_really_inline constexpr void atom(W &w, const T &container) {
   if (container.empty()) {
-    b.append_raw("[]");
+    if (!w.ensure(2)) return;
+    std::memcpy(w.ptr + w.pos, "[]", 2);
+    w.pos += 2;
     return;
   }
-  b.append('[');
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = '[';
   bool first = true;
   for (const auto& item : container) {
     if (!first) {
-      b.append(',');
+      if (!w.ensure(1)) return;
+      w.ptr[w.pos++] = ',';
     }
     first = false;
-    atom(b, item);
+    atom(w, item);
+  }
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = ']';
+}
+
+// =============================================================
+// Size bound: an upper bound on the number of bytes that atom(w, t) writes.
+// Computing it first lets append() reserve the capacity once and then run
+// the whole write chain through an unchecked_writer, without a capacity
+// check before every write. It mirrors the atom() overloads above.
+//
+// Prior related work: jsonifier sizes the document first, counting 6 bytes
+// per string byte, resizes once to that bound plus slack, and writes with
+// no capacity check on each store. to_json() does that resize through
+// std::string::resize_and_overwrite. See serializer.hpp and
+// serialize_impl.hpp in https://github.com/nihilai-collective/Jsonifier
+// and https://nihilai-collective.net/serialization.
+// =============================================================
+namespace bound_detail {
+
+// Whether size_bound covers everything that atom() writes for T: not when a
+// member is serialized by a with<Adapter> serializer, which writes an unknown
+// amount through the string_builder.
+template <class T>
+consteval bool is_bounded() {
+  if constexpr (require_custom_serialization<T>) {
+    return false;
+  } else if constexpr (std::is_same_v<T, std::string> || std::is_same_v<T, std::string_view> ||
+                       std::is_same_v<T, const char *> || std::is_arithmetic_v<T> || std::is_enum_v<T>) {
+    return true;
+  } else if constexpr (concepts::optional_type<T> || concepts::smart_pointer<T>) {
+    return is_bounded<std::remove_cvref_t<decltype(*std::declval<const T &>())>>();
+  } else if constexpr (concepts::string_view_keyed_map<T>) {
+    return is_bounded<std::remove_cvref_t<typename T::mapped_type>>();
+  } else if constexpr (concepts::container_but_not_string<T> || concepts::appendable_containers<T>) {
+    return is_bounded<std::remove_cvref_t<std::ranges::range_value_t<T>>>();
+  } else {
+    bool bounded = true;
+    template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+      if constexpr (annotation_detail::is_serialized_member(dm)) {
+        bounded = bounded && simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::with_t) == std::meta::info{} &&
+                  is_bounded<std::remove_cvref_t<decltype(std::declval<const T &>().[:dm:])>>();
+      }
+    };
+    return bounded;
+  }
+}
+
+template <class T>
+consteval size_t enum_bound() {
+  size_t bound = 20; // the integer fallback
+  template for (constexpr auto enum_val : std::define_static_array(std::meta::enumerators_of(^^T))) {
+    constexpr size_t len = std::char_traits<char>::length(std::define_static_string(
+        constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<enum_val>())));
+    bound = (std::max)(bound, len);
+  };
+  return bound;
+}
+
+template <class T>
+simdjson_really_inline size_t size_bound(const T &t) noexcept;
+
+// Bound for the "key":value pairs of a structure, commas included.
+template <class T>
+simdjson_really_inline size_t fields_bound(const T &t) noexcept {
+  size_t bound = 0;
+  template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+    if constexpr (annotation_detail::is_serialized_member(dm)) {
+      if constexpr (simdjson::detail::has_annotation(dm, ^^simdjson::detail::flatten_tag)) {
+        bound += fields_bound(t.[:dm:]);
+      } else {
+        constexpr size_t rest_key_len = constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<dm>()).size() + 2;
+        bound += rest_key_len + size_bound(t.[:dm:]);
+      }
+    }
+  };
+  return bound;
+}
+
+template <class T>
+simdjson_really_inline size_t size_bound([[maybe_unused]] const T &t) noexcept {
+  if constexpr (std::is_same_v<T, char>) {
+    return 2 + 6;
+  } else if constexpr (std::is_same_v<T, std::string> || std::is_same_v<T, std::string_view> ||
+                       std::is_same_v<T, const char *>) {
+    // Every byte may become \uXXXX, plus the quotes.
+    return 2 + 6 * std::string_view(t).size();
+  } else if constexpr (std::is_same_v<T, bool>) {
+    return 5;
+  } else if constexpr (std::is_floating_point_v<T>) {
+    return simdjson::internal::to_chars_buffer_size;
+  } else if constexpr (std::is_arithmetic_v<T>) {
+    return 20;
+  } else if constexpr (std::is_enum_v<T>) {
+    return enum_bound<T>();
+  } else if constexpr (concepts::optional_type<T> || concepts::smart_pointer<T>) {
+    return t ? size_bound(*t) : 4;
+  } else if constexpr (concepts::string_view_keyed_map<T>) {
+    size_t bound = 2;
+    for (const auto &[key, value] : t) {
+      // comma, quotes, colon
+      bound += 4 + 6 * std::string_view(key).size() + size_bound(value);
+    }
+    return bound;
+  } else if constexpr (concepts::container_but_not_string<T> || concepts::appendable_containers<T>) {
+    using value_type = std::remove_cvref_t<std::ranges::range_value_t<T>>;
+    if constexpr (std::is_arithmetic_v<value_type> && !std::is_same_v<value_type, char>) {
+      // A fixed bound per element: no need to visit them.
+      return 2 + size_t(std::ranges::distance(t)) * (1 + size_bound(value_type{}));
+    } else {
+      size_t bound = 2;
+#if defined(__GNUC__) && !defined(__clang__)
+#pragma GCC novector // a vector loop is slower on short containers
+#endif
+#pragma GCC unroll 4
+      for (const auto &item : t) {
+        bound += 1 + size_bound(item);
+      }
+      return bound;
+    }
+  } else if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag)) {
+    constexpr auto dm = simdjson::detail::transparent_member(^^T);
+    return size_bound(t.[:dm:]);
+  } else {
+    return 2 + fields_bound(t);
+  }
+}
+
+} // namespace bound_detail
+
+// Write t through an unchecked writer when its size bound is available,
+// reserving that many bytes first, and through the checked writer otherwise.
+template <class T>
+simdjson_really_inline void append_bounded(string_builder &b, const T &t) {
+  // On 32-bit systems, the bound could overflow: keep the checked writer.
+  if constexpr (sizeof(size_t) >= 8 && bound_detail::is_bounded<T>()) {
+    const size_t bound = bound_detail::size_bound(t) + unchecked_slack;
+    const size_t pos = b.unsafe_position();
+    // The bound is a sum of in-memory sizes times a small constant: it cannot
+    // overflow on a 64-bit system. Be pedantic elsewhere.
+    if (sizeof(size_t) >= 8 || bound <= (std::numeric_limits<size_t>::max)() - pos) {
+      const size_t cap = b.unsafe_capacity();
+      // Grow geometrically so that many small appends stay amortized.
+      if (pos + bound <= cap || b.unsafe_grow((std::max)(cap * 2, pos + bound))) {
+        unchecked_writer w(b.unsafe_data(), pos);
+        atom(w, t);
+        b.unsafe_set_position(w.pos);
+      }
+      return;
+    }
   }
-  b.append(']');
+  writer w(b);
+  atom(w, t);
+  w.sync();
 }

-// append functions that delegate to atom functions for primitive types
+// append() -- top-level entry. Each overload constructs a stack-local
+// writer, runs atom(w, t) through the inlined call chain, then syncs
+// the local position back into the string_builder.
 template <class T>
   requires(std::is_arithmetic_v<T> && !std::is_same_v<T, char>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  writer w(b);
+  atom(w, t);
+  w.sync();
 }

 template <class T>
@@ -57521,20 +70504,22 @@ template <class T>
            std::is_same_v<T, std::string_view> ||
            std::is_same_v<T, const char *> ||
            std::is_same_v<T, char>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  writer w(b);
+  atom(w, t);
+  w.sync();
 }

 template <concepts::optional_type T>
   requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  append_bounded(b, t);
 }

 template <concepts::smart_pointer T>
   requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  append_bounded(b, t);
 }

 template <concepts::appendable_containers T>
@@ -57542,14 +70527,14 @@ template <concepts::appendable_containers T>
            !concepts::optional_type<T> && !concepts::smart_pointer<T> &&
            !std::is_same_v<T, std::string> &&
            !std::is_same_v<T, std::string_view> && !std::is_same_v<T, const char*> && !require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  append_bounded(b, t);
 }

 template <concepts::string_view_keyed_map T>
   requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  append_bounded(b, t);
 }

 // works for struct
@@ -57563,39 +70548,15 @@ template <class Z>
            !std::is_same_v<Z, std::string_view> &&
            !std::is_same_v<Z, const char*> &&
            !std::is_same_v<Z, char> && !require_custom_serialization<Z>)
-void append(string_builder &b, const Z &z) {
-  int i = 0;
-  b.append('{');
-  template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^Z, std::meta::access_context::unchecked()))) {
-    if (i != 0)
-      b.append(',');
-    constexpr auto key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(dm)));
-    b.append_raw(key);
-    b.append(':');
-    atom(b, z.[:dm:]);
-    i++;
-  };
-  b.append('}');
+simdjson_inline void append(string_builder &b, const Z &z) {
+  append_bounded(b, z);
 }

 // works for container that have begin() and end() iterators
 template <class Z>
   requires(concepts::container_but_not_string<Z> && !require_custom_serialization<Z>)
-void append(string_builder &b, const Z &z) {
-  auto it = z.begin();
-  auto end = z.end();
-  if (it == end) {
-    b.append_raw("[]");
-    return;
-  }
-  b.append('[');
-  atom(b, *it);
-  ++it;
-  for (; it != end; ++it) {
-    b.append(',');
-    atom(b, *it);
-  }
-  b.append(']');
+simdjson_inline void append(string_builder &b, const Z &z) {
+  append_bounded(b, z);
 }

 template <class Z>
@@ -57606,22 +70567,40 @@ void append(string_builder &b, const Z &z) {


 template <class Z>
-simdjson_warn_unused simdjson_result<std::string> to_json_string(const Z &z, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
-  string_builder b(initial_capacity);
-  append(b, z);
-  std::string_view s;
-  if(auto e = b.view().get(s); e) { return e; }
-  return std::string(s);
+simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
+  if constexpr (sizeof(size_t) >= 8 && bound_detail::is_bounded<Z>()) {
+    // Write straight into s, sized by the bound: no intermediate buffer, no copy.
+    // Prior related work: jsonifier's serializeJson resizes once through
+    // resize_and_overwrite (serializer.hpp).
+    (void)initial_capacity;
+    const size_t bound = bound_detail::size_bound(z) + unchecked_slack;
+    auto write = [&z](char *p) noexcept {
+      unchecked_writer w(p, 0);
+      atom(w, z);
+      return w.pos;
+    };
+#if defined(__cpp_lib_string_resize_and_overwrite) && __cpp_lib_string_resize_and_overwrite >= 202110L
+    s.resize_and_overwrite(bound, [&write](char *p, size_t) noexcept { return write(p); });
+#else
+    s.resize(bound);
+    s.resize(write(s.data()));
+#endif
+    return SUCCESS;
+  } else {
+    string_builder b(initial_capacity);
+    append(b, z);
+    std::string_view view;
+    if(auto e = b.view().get(view); e) { return e; }
+    s.assign(view);
+    return SUCCESS;
+  }
 }

 template <class Z>
-simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
-  string_builder b(initial_capacity);
-  append(b, z);
-  std::string_view view;
-  if(auto e = b.view().get(view); e) { return e; }
-  s.assign(view);
-  return SUCCESS;
+simdjson_warn_unused simdjson_result<std::string> to_json_string(const Z &z, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
+  std::string s;
+  if(auto e = to_json(z, s, initial_capacity); e) { return e; }
+  return s;
 }

 template <class Z>
@@ -57634,40 +70613,41 @@ string_builder& operator<<(string_builder& b, const Z& z) {
 template<constevalutil::fixed_string... FieldNames, typename T>
   requires(std::is_class_v<T> && (sizeof...(FieldNames) > 0))
 void extract_from(string_builder &b, const T &obj) {
-  // Helper to check if a field name matches any of the requested fields
-  auto should_extract = [](std::string_view field_name) constexpr -> bool {
-    return ((FieldNames.view() == field_name) || ...);
-  };
-
-  b.append('{');
+  writer w(b);
+  if (!w.ensure(1)) { w.sync(); return; }
+  w.ptr[w.pos++] = '{';
   bool first = true;
-
   // Iterate through all members of T using reflection
-  template for (constexpr auto mem : std::define_static_array(
-      std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
-
+  static constexpr auto members = std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()));
+  template for (constexpr auto mem : members) {
     if constexpr (std::meta::is_public(mem)) {
-      constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
+      static constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));

       // Only serialize this field if it's in our list of requested fields
-      if constexpr (should_extract(key)) {
-        if (!first) {
-          b.append(',');
+      if constexpr (((FieldNames.view() == key) || ...)) {
+        static constexpr auto first_key = std::define_static_string(
+            constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)) + ":");
+        static constexpr auto rest_key = std::define_static_string(
+            std::string(",") + constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)) + ":");
+        constexpr size_t first_key_len = std::char_traits<char>::length(first_key);
+        constexpr size_t rest_key_len = std::char_traits<char>::length(rest_key);
+        if (!w.ensure(rest_key_len)) { w.sync(); return; }
+        if (first) {
+          std::memcpy(w.ptr + w.pos, first_key, first_key_len);
+          w.pos += first_key_len;
+        } else {
+          std::memcpy(w.ptr + w.pos, rest_key, rest_key_len);
+          w.pos += rest_key_len;
         }
         first = false;
-
-        // Serialize the key
-        constexpr auto quoted_key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)));
-        b.append_raw(quoted_key);
-        b.append(':');
-
-        // Serialize the value
-        atom(b, obj.[:mem:]);
+        atom(w, obj.[:mem:]);
       }
     }
   };

-  b.append('}');
+  if (!w.ensure(1)) { w.sync(); return; }
+  w.ptr[w.pos++] = '}';
+  w.sync();
 }

 template<constevalutil::fixed_string... FieldNames, typename T>
@@ -57680,25 +70660,18 @@ simdjson_warn_unused simdjson_result<std::string> extract_from(const T &obj, siz
   return std::string(s);
 }

+SIMDJSON_POP_DISABLE_WARNINGS
+
 } // namespace builder
 } // namespace lasx
 // Alias the function template to 'to' in the global namespace
 template <class Z>
 simdjson_warn_unused simdjson_result<std::string> to_json(const Z &z, size_t initial_capacity = lasx::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
-  lasx::builder::string_builder b(initial_capacity);
-  lasx::builder::append(b, z);
-  std::string_view s;
-  if(auto e = b.view().get(s); e) { return e; }
-  return std::string(s);
+  return lasx::builder::to_json_string(z, initial_capacity);
 }
 template <class Z>
 simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = lasx::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
-  lasx::builder::string_builder b(initial_capacity);
-  lasx::builder::append(b, z);
-  std::string_view view;
-  if(auto e = b.view().get(view); e) { return e; }
-  s.assign(view);
-  return SUCCESS;
+  return lasx::builder::to_json(z, s, initial_capacity);
 }
 // Global namespace function for extract_from
 template<constevalutil::fixed_string... FieldNames, typename T>
@@ -57844,6 +70817,7 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
 /* including simdjson/generic/builder/json_string_builder-inl.h for lasx: #include "simdjson/generic/builder/json_string_builder-inl.h" */
 /* begin file simdjson/generic/builder/json_string_builder-inl.h for lasx */
 #include <array>
+#include <cmath>
 #include <cstring>
 #include <limits>
 #include <type_traits>
@@ -57878,6 +70852,11 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
 #define SIMDJSON_EXPERIMENTAL_HAS_LSX 1
 #endif
 #endif
+#if defined(__loongarch_asx)
+#ifndef SIMDJSON_EXPERIMENTAL_HAS_LASX
+#define SIMDJSON_EXPERIMENTAL_HAS_LASX 1
+#endif
+#endif
 #if defined(__riscv_v_intrinsic) && __riscv_v_intrinsic >= 11000 &&            \
     defined(__riscv_vector)
 #ifndef SIMDJSON_EXPERIMENTAL_HAS_RVV
@@ -57897,6 +70876,9 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
 #endif
 #if SIMDJSON_EXPERIMENTAL_HAS_SSE2
 #include <emmintrin.h>
+#if defined(__AVX2__)
+#include <immintrin.h>
+#endif
 #ifdef _MSC_VER
 #include <intrin.h>
 #endif
@@ -57904,6 +70886,9 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
 #if SIMDJSON_EXPERIMENTAL_HAS_LSX
 #include <lsxintrin.h>
 #endif
+#if SIMDJSON_EXPERIMENTAL_HAS_LASX
+#include <lasxintrin.h>
+#endif
 #if SIMDJSON_EXPERIMENTAL_HAS_RVV
 #include <riscv_vector.h>
 #endif
@@ -57953,105 +70938,6 @@ inline bool has_json_escapable_byte(uint64_t x) {

 **/

-SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline bool
-simple_needs_escaping(std::string_view v) {
-  for (char c : v) {
-    // a table lookup is faster than a series of comparisons
-    if (json_quotable_character[static_cast<uint8_t>(c)]) {
-      return true;
-    }
-  }
-  return false;
-}
-
-#if SIMDJSON_EXPERIMENTAL_HAS_NEON
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
-  if (view.size() < 16) {
-    return simple_needs_escaping(view);
-  }
-  size_t i = 0;
-  uint8x16_t running = vdupq_n_u8(0);
-  uint8x16_t v34 = vdupq_n_u8(34);
-  uint8x16_t v92 = vdupq_n_u8(92);
-
-  for (; i + 15 < view.size(); i += 16) {
-    uint8x16_t word = vld1q_u8((const uint8_t *)view.data() + i);
-    running = vorrq_u8(running, vceqq_u8(word, v34));
-    running = vorrq_u8(running, vceqq_u8(word, v92));
-    running = vorrq_u8(running, vcltq_u8(word, vdupq_n_u8(32)));
-  }
-  if (i < view.size()) {
-    uint8x16_t word =
-        vld1q_u8((const uint8_t *)view.data() + view.length() - 16);
-    running = vorrq_u8(running, vceqq_u8(word, v34));
-    running = vorrq_u8(running, vceqq_u8(word, v92));
-    running = vorrq_u8(running, vcltq_u8(word, vdupq_n_u8(32)));
-  }
-  return vmaxvq_u32(vreinterpretq_u32_u8(running)) != 0;
-}
-#elif SIMDJSON_EXPERIMENTAL_HAS_SSE2
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
-  if (view.size() < 16) {
-    return simple_needs_escaping(view);
-  }
-  size_t i = 0;
-  __m128i running = _mm_setzero_si128();
-  for (; i + 15 < view.size(); i += 16) {
-
-    __m128i word =
-        _mm_loadu_si128(reinterpret_cast<const __m128i *>(view.data() + i));
-    running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(34)));
-    running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(92)));
-    running = _mm_or_si128(
-        running, _mm_cmpeq_epi8(_mm_subs_epu8(word, _mm_set1_epi8(31)),
-                                _mm_setzero_si128()));
-  }
-  if (i < view.size()) {
-    __m128i word = _mm_loadu_si128(
-        reinterpret_cast<const __m128i *>(view.data() + view.length() - 16));
-    running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(34)));
-    running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(92)));
-    running = _mm_or_si128(
-        running, _mm_cmpeq_epi8(_mm_subs_epu8(word, _mm_set1_epi8(31)),
-                                _mm_setzero_si128()));
-  }
-  return _mm_movemask_epi8(running) != 0;
-}
-#elif SIMDJSON_EXPERIMENTAL_HAS_PPC64
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
-  if (view.size() < 16) {
-    return simple_needs_escaping(view);
-  }
-  size_t i = 0;
-  __vector unsigned char running = vec_splats((unsigned char)0);
-  __vector unsigned char v34 = vec_splats((unsigned char)34);
-  __vector unsigned char v92 = vec_splats((unsigned char)92);
-  __vector unsigned char v32 = vec_splats((unsigned char)32);
-
-  for (; i + 15 < view.size(); i += 16) {
-    __vector unsigned char word =
-        vec_vsx_ld(0, reinterpret_cast<const unsigned char *>(view.data() + i));
-    running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v34));
-    running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v92));
-    running = vec_or(running,
-        (__vector unsigned char)vec_cmplt(word, v32));
-  }
-  if (i < view.size()) {
-    __vector unsigned char word = vec_vsx_ld(
-        0, reinterpret_cast<const unsigned char *>(view.data() + view.length() - 16));
-    running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v34));
-    running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v92));
-    running = vec_or(running,
-        (__vector unsigned char)vec_cmplt(word, v32));
-  }
-  return !vec_all_eq(running, vec_splats((unsigned char)0));
-}
-#else
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
-  return simple_needs_escaping(view);
-}
-#endif
-
 // Scalar fallback for finding next quotable character
 SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline size_t
 find_next_json_quotable_character_scalar(const std::string_view view,
@@ -58142,6 +71028,51 @@ find_next_json_quotable_character(const std::string_view view,
   size_t current = len - remaining;
   return find_next_json_quotable_character_scalar(view, current);
 }
+#elif SIMDJSON_EXPERIMENTAL_HAS_LASX
+simdjson_inline size_t
+find_next_json_quotable_character(const std::string_view view,
+                                  size_t location) noexcept {
+  const size_t len = view.size();
+  const uint8_t *ptr =
+      reinterpret_cast<const uint8_t *>(view.data()) + location;
+  size_t remaining = len - location;
+
+  // SIMD constants for characters requiring escape
+  __m256i v34 = __lasx_xvreplgr2vr_b(34);  // '"'
+  __m256i v92 = __lasx_xvreplgr2vr_b(92);  // '\\'
+  __m256i v32 = __lasx_xvreplgr2vr_b(32);  // control char threshold
+
+  while (remaining >= 32) {
+    __m256i word = __lasx_xvld(ptr, 0);
+
+    // Check for quotable characters: '"', '\\', or control chars (< 32)
+    __m256i needs_escape = __lasx_xvseq_b(word, v34);
+    needs_escape = __lasx_xvor_v(needs_escape, __lasx_xvseq_b(word, v92));
+    needs_escape = __lasx_xvor_v(needs_escape, __lasx_xvslt_bu(word, v32));
+
+    if (!__lasx_xbz_v(needs_escape)) {
+      // Found a quotable character - locate it via the four 64-bit lanes
+      uint64_t lane0 = __lasx_xvpickve2gr_du(needs_escape, 0);
+      uint64_t lane1 = __lasx_xvpickve2gr_du(needs_escape, 1);
+      uint64_t lane2 = __lasx_xvpickve2gr_du(needs_escape, 2);
+      uint64_t lane3 = __lasx_xvpickve2gr_du(needs_escape, 3);
+      size_t offset = ptr - reinterpret_cast<const uint8_t *>(view.data());
+      if (lane0 != 0) {
+        return offset + trailing_zeroes(lane0) / 8;
+      } else if (lane1 != 0) {
+        return offset + 8 + trailing_zeroes(lane1) / 8;
+      } else if (lane2 != 0) {
+        return offset + 16 + trailing_zeroes(lane2) / 8;
+      } else {
+        return offset + 24 + trailing_zeroes(lane3) / 8;
+      }
+    }
+    ptr += 32;
+    remaining -= 32;
+  }
+  size_t current = len - remaining;
+  return find_next_json_quotable_character_scalar(view, current);
+}
 #elif SIMDJSON_EXPERIMENTAL_HAS_LSX
 simdjson_inline size_t
 find_next_json_quotable_character(const std::string_view view,
@@ -58297,6 +71228,254 @@ SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline void escape_json_char(char c, char *&o
   }
 }

+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2 || SIMDJSON_EXPERIMENTAL_HAS_NEON
+#define SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE 1
+#endif
+
+#if SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2
+
+using escape_vector = __m128i;
+// Mask bits that each input byte contributes to escape_bitmask().
+static constexpr unsigned escape_mask_bits = 1;
+
+simdjson_inline escape_vector escape_load16(const uint8_t *p) noexcept {
+  return _mm_loadu_si128(reinterpret_cast<const __m128i *>(p));
+}
+
+simdjson_inline void escape_store16(char *out, escape_vector v) noexcept {
+  _mm_storeu_si128(reinterpret_cast<__m128i *>(out), v);
+}
+
+// Builds a vector whose bytes 0..7 come from a and bytes 8..15 from b.
+simdjson_inline escape_vector escape_load8x2(const uint8_t *a,
+                                             const uint8_t *b) noexcept {
+  return _mm_unpacklo_epi64(
+      _mm_loadl_epi64(reinterpret_cast<const __m128i *>(a)),
+      _mm_loadl_epi64(reinterpret_cast<const __m128i *>(b)));
+}
+
+// Builds a vector whose bytes 0..3 come from a and bytes 4..7 from b. The
+// upper half is unspecified; callers only look at the low eight bits.
+simdjson_inline escape_vector escape_load4x2(const uint8_t *a,
+                                             const uint8_t *b) noexcept {
+  int32_t a32, b32;
+  memcpy(&a32, a, 4);
+  memcpy(&b32, b, 4);
+  return _mm_unpacklo_epi32(_mm_cvtsi32_si128(a32), _mm_cvtsi32_si128(b32));
+}
+
+// Sets every bit of byte k when byte k of v is a quotable character ('"',
+// '\\' or a control character).
+simdjson_inline escape_vector escape_flags(escape_vector v) noexcept {
+  const __m128i v34 = _mm_set1_epi8(34); // '"'
+  const __m128i v92 = _mm_set1_epi8(92); // '\\'
+  const __m128i v31 = _mm_set1_epi8(31); // for control char detection
+  __m128i needs_escape = _mm_cmpeq_epi8(v, v34);
+  needs_escape = _mm_or_si128(needs_escape, _mm_cmpeq_epi8(v, v92));
+  return _mm_or_si128(
+      needs_escape, _mm_cmpeq_epi8(_mm_subs_epu8(v, v31), _mm_setzero_si128()));
+}
+
+// True when any byte needs escaping. Kept separate from escape_bitmask
+// because some instruction sets can answer it without leaving the vector
+// register file; on SSE2 the compiler folds the two together.
+simdjson_inline bool escape_any(escape_vector flags) noexcept {
+  return _mm_movemask_epi8(flags) != 0;
+}
+
+// A 16-bit mask with bit k set when byte k needed escaping.
+simdjson_inline uint64_t escape_bitmask(escape_vector flags) noexcept {
+  return uint64_t(uint32_t(_mm_movemask_epi8(flags)));
+}
+
+#else // SIMDJSON_EXPERIMENTAL_HAS_NEON
+
+using escape_vector = uint8x16_t;
+static constexpr unsigned escape_mask_bits = 4;
+
+simdjson_inline escape_vector escape_load16(const uint8_t *p) noexcept {
+  return vld1q_u8(p);
+}
+
+simdjson_inline void escape_store16(char *out, escape_vector v) noexcept {
+  vst1q_u8(reinterpret_cast<uint8_t *>(out), v);
+}
+
+simdjson_inline escape_vector escape_load8x2(const uint8_t *a,
+                                             const uint8_t *b) noexcept {
+  uint64_t a64, b64;
+  memcpy(&a64, a, 8);
+  memcpy(&b64, b, 8);
+  return vreinterpretq_u8_u64(vsetq_lane_u64(b64, vdupq_n_u64(a64), 1));
+}
+
+simdjson_inline escape_vector escape_load4x2(const uint8_t *a,
+                                             const uint8_t *b) noexcept {
+  uint32_t a32, b32;
+  memcpy(&a32, a, 4);
+  memcpy(&b32, b, 4);
+  return vreinterpretq_u8_u32(vsetq_lane_u32(b32, vdupq_n_u32(a32), 1));
+}
+
+simdjson_inline escape_vector escape_flags(escape_vector v) noexcept {
+  uint8x16_t needs_escape = vceqq_u8(v, vdupq_n_u8(34));              // '"'
+  needs_escape = vorrq_u8(needs_escape, vceqq_u8(v, vdupq_n_u8(92))); // '\\'
+  return vorrq_u8(needs_escape, vcltq_u8(v, vdupq_n_u8(32)));
+}
+
+simdjson_inline bool escape_any(escape_vector flags) noexcept {
+  return vmaxvq_u32(vreinterpretq_u32_u8(flags)) != 0;
+}
+
+// Four bits per byte rather than one: escape_block and the tail paths scale
+// their shifts by escape_mask_bits to match.
+simdjson_inline uint64_t escape_bitmask(escape_vector flags) noexcept {
+  uint8x8_t narrowed = vshrn_n_u16(vreinterpretq_u16_u8(flags), 4);
+  return vget_lane_u64(vreinterpret_u64_u8(narrowed), 0);
+}
+
+#endif // instruction set selection
+
+// The escape bitmask of a 16-byte block.
+simdjson_inline uint64_t escape_mask(escape_vector v) noexcept {
+  return escape_bitmask(escape_flags(v));
+}
+
+// Copies n bytes with n < 16, using overlapping loads and stores. It never
+// reads more than n bytes from src, nor writes more than n bytes to dst.
+simdjson_inline void copy_lt16(char *dst, const uint8_t *src,
+                               size_t n) noexcept {
+  if (n >= 8) {
+    memcpy(dst, src, 8);
+    memcpy(dst + n - 8, src + n - 8, 8);
+  } else if (n >= 4) {
+    memcpy(dst, src, 4);
+    memcpy(dst + n - 4, src + n - 4, 4);
+  } else if (n > 0) {
+    dst[0] = char(src[0]);
+    dst[n >> 1] = char(src[n >> 1]);
+    dst[n - 1] = char(src[n - 1]);
+  }
+}
+
+// Escapes the bytes of src in the range [i, blockend), given that m is the
+// (non-zero) escape mask for that range: the escape_mask_bits-wide lane of m
+// at byte k is non-zero when src[i + k] requires escaping. Returns the updated
+// output pointer.
+simdjson_never_inline char *escape_block(const uint8_t *src, char *out,
+                                         size_t i, size_t blockend,
+                                         uint64_t m) noexcept {
+  constexpr uint64_t lane = (uint64_t(1) << escape_mask_bits) - 1;
+
+  size_t pos = i; // first byte not yet copied
+  while (m) {
+    const size_t tz = trailing_zeroes(m);
+    const size_t next = i + tz / escape_mask_bits;
+    // Copy the run of safe bytes that precedes this escape.
+    copy_lt16(out, src + pos, next - pos);
+    out += next - pos;
+    escape_json_char(char(src[next]), out);
+    pos = next + 1;
+    m &= ~(lane << tz);
+  }
+  // Copy whatever follows the last escape.
+  copy_lt16(out, src + pos, blockend - pos);
+  return out + (blockend - pos);
+}
+
+// Writes the escaped version of input to out, returning the number of bytes
+// written.
+simdjson_really_inline size_t write_string_escaped(const std::string_view input, char *out) {
+  const size_t len = input.size();
+  const uint8_t *src = reinterpret_cast<const uint8_t *>(input.data());
+  const char *const initout = out;
+
+  size_t i = 0;
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2 && defined(__AVX2__)
+  while (i + 32 <= len) {
+    const __m256i word = _mm256_loadu_si256(reinterpret_cast<const __m256i *>(src + i));
+    const __m256i flags = _mm256_or_si256(
+        _mm256_or_si256(_mm256_cmpeq_epi8(word, _mm256_set1_epi8(34)),   // '"'
+                        _mm256_cmpeq_epi8(word, _mm256_set1_epi8(92))),  // '\\'
+        _mm256_cmpeq_epi8(_mm256_subs_epu8(word, _mm256_set1_epi8(31)),
+                          _mm256_setzero_si256()));                      // control
+    const uint32_t mask = uint32_t(_mm256_movemask_epi8(flags));
+    if (simdjson_likely(mask == 0)) {
+      _mm256_storeu_si256(reinterpret_cast<__m256i *>(out), word);
+      out += 32;
+    } else {
+      for (size_t half = 0; half < 32; half += 16) {
+        const uint64_t m = (mask >> half) & 0xFFFF;
+        if (m == 0) {
+          escape_store16(out, escape_load16(src + i + half));
+          out += 16;
+        } else {
+          out = escape_block(src, out, i + half, i + half + 16, m);
+        }
+      }
+    }
+    i += 32;
+  }
+#endif
+  while (i + 16 <= len) {
+    escape_vector word = escape_load16(src + i);
+    escape_vector flags = escape_flags(word);
+    if (simdjson_likely(!escape_any(flags))) {
+      escape_store16(out, word);
+      out += 16;
+    } else {
+      out = escape_block(src, out, i, i + 16, escape_bitmask(flags));
+    }
+    i += 16;
+  }
+  if (i < len) {
+    const size_t rem = len - i;
+    uint64_t m;
+    if (len >= 16) {
+      // The last 16 bytes of the input are in bounds. Bit k of that block's
+      // mask belongs to input position len - 16 + k, so shift it down to align
+      // bit 0 with position i.
+      m = escape_mask(escape_load16(src + len - 16)) >>
+          (escape_mask_bits * (16 - rem));
+    } else if (len >= 8) {
+      // Here i == 0 and rem == len. Two overlapping 8-byte loads cover
+      // [0, 8) and [len - 8, len), which is the whole input since len < 16.
+      uint64_t mm = escape_mask(escape_load8x2(src, src + len - 8));
+      constexpr uint64_t low8 = (uint64_t(1) << (escape_mask_bits * 8)) - 1;
+      m = (mm & low8) |
+          ((mm >> (escape_mask_bits * 8)) << (escape_mask_bits * (len - 8)));
+    } else if (len >= 4) {
+      // Same idea with two overlapping 4-byte loads.
+      uint64_t mm = escape_mask(escape_load4x2(src, src + len - 4));
+      constexpr uint64_t low4 = (uint64_t(1) << (escape_mask_bits * 4)) - 1;
+      m = (mm & low4) | (((mm >> (escape_mask_bits * 4)) & low4)
+                         << (escape_mask_bits * (len - 4)));
+    } else {
+      // Fewer than 4 bytes: at most three table lookups, no need for SIMD.
+      for (size_t k = 0; k < len; k++) {
+        uint8_t c = src[k];
+        if (json_quotable_character[c]) {
+          escape_json_char(char(c), out);
+        } else {
+          *out++ = char(c);
+        }
+      }
+      return size_t(out - initout);
+    }
+    if (m == 0) {
+      copy_lt16(out, src + i, rem);
+      out += rem;
+    } else {
+      out = escape_block(src, out, i, len, m);
+    }
+  }
+  return size_t(out - initout);
+}
+
+#else // SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
 // Writes the escaped version of input to out, returning the number of bytes
 // written. Uses SIMD position finding to locate quotable characters efficiently.
 inline size_t write_string_escaped(const std::string_view input, char *out) {
@@ -58326,9 +71505,13 @@ inline size_t write_string_escaped(const std::string_view input, char *out) {
     escape_json_char(input[location], out);
     location += 1;
   }
-  return out - initout;
+  return size_t(out - initout);
 }

+#endif // SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+#undef SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
+
 simdjson_inline string_builder::string_builder(size_t initial_capacity)
     : buffer(new(std::nothrow) char[initial_capacity]), position(0),
       capacity(buffer.get() != nullptr ? initial_capacity : 0),
@@ -58351,7 +71534,7 @@ simdjson_inline bool string_builder::capacity_check(size_t upcoming_bytes) {
   return is_valid;
 }

-simdjson_inline void string_builder::grow_buffer(size_t desired_capacity) {
+inline void string_builder::grow_buffer(size_t desired_capacity) {
   if (!is_valid) {
     return;
   }
@@ -58407,81 +71590,136 @@ simdjson_inline void string_builder::clear() noexcept {

 namespace internal {

-template <typename number_type, typename = typename std::enable_if<
-                                    std::is_unsigned<number_type>::value>::type>
-simdjson_really_inline int int_log2(number_type x) {
-  return 63 - leading_zeroes(uint64_t(x) | 1);
-}
-
-simdjson_really_inline int fast_digit_count_32(uint32_t x) {
-  static uint64_t table[] = {
-      4294967296,  8589934582,  8589934582,  8589934582,  12884901788,
-      12884901788, 12884901788, 17179868184, 17179868184, 17179868184,
-      21474826480, 21474826480, 21474826480, 21474826480, 25769703776,
-      25769703776, 25769703776, 30063771072, 30063771072, 30063771072,
-      34349738368, 34349738368, 34349738368, 34349738368, 38554705664,
-      38554705664, 38554705664, 41949672960, 41949672960, 41949672960,
-      42949672960, 42949672960};
-  return uint32_t((x + table[int_log2(x)]) >> 32);
-}
-
-simdjson_really_inline int fast_digit_count_64(uint64_t x) {
-  static uint64_t table[] = {9,
-                             99,
-                             999,
-                             9999,
-                             99999,
-                             999999,
-                             9999999,
-                             99999999,
-                             999999999,
-                             9999999999,
-                             99999999999,
-                             999999999999,
-                             9999999999999,
-                             99999999999999,
-                             999999999999999ULL,
-                             9999999999999999ULL,
-                             99999999999999999ULL,
-                             999999999999999999ULL,
-                             9999999999999999999ULL};
-  int y = (19 * int_log2(x) >> 6);
-  y += x > table[y];
-  return y + 1;
-}
-
-template <typename number_type, typename = typename std::enable_if<
-                                    std::is_unsigned<number_type>::value>::type>
-simdjson_really_inline size_t digit_count(number_type v) noexcept {
-  static_assert(sizeof(number_type) == 8 || sizeof(number_type) == 4 ||
-                    sizeof(number_type) == 2 || sizeof(number_type) == 1,
-                "We only support 8-bit, 16-bit, 32-bit and 64-bit numbers");
-  SIMDJSON_IF_CONSTEXPR(sizeof(number_type) <= 4) {
-    return fast_digit_count_32(static_cast<uint32_t>(v));
+// Integer to decimal: James Edward Anhalt III's algorithm
+static const char jeaiii_dd[201] =
+    "00010203040506070809101112131415161718192021222324252627282930313233343536373839"
+    "40414243444546474849505152535455565758596061626364656667686970717273747576777879"
+    "8081828384858687888990919293949596979899";
+static const char jeaiii_fd[201] =
+    "0\0" "1\0" "2\0" "3\0" "4\0" "5\0" "6\0" "7\0" "8\0" "9\0"
+    "10111213141516171819202122232425262728293031323334353637383940414243444546474849"
+    "50515253545556575859606162636465666768697071727374757677787980818283848586878889"
+    "90919293949596979899";
+
+simdjson_really_inline void jeaiii_write_dd(char *p, uint64_t k) noexcept {
+  std::memcpy(p, &jeaiii_dd[2 * k], 2);
+}
+simdjson_really_inline void jeaiii_write_fd(char *p, uint64_t k) noexcept {
+  std::memcpy(p, &jeaiii_fd[2 * k], 2);
+}
+
+// Caller guarantees n < 10^8. Writes 1 to 8 digits.
+simdjson_really_inline char *jeaiii_lt1e8(char *b, uint32_t n) noexcept {
+  constexpr uint64_t mask24 = (uint64_t(1) << 24) - 1;
+  constexpr uint64_t mask32 = (uint64_t(1) << 32) - 1;
+  if (n < 100) {
+    jeaiii_write_fd(b, n);
+    return n < 10 ? b + 1 : b + 2;
+  }
+  if (n < 1000000) {
+    if (n < 10000) {
+      const uint32_t f0 = uint32_t(10 * (1 << 24) / 1e3 + 1) * n;
+      jeaiii_write_fd(b, f0 >> 24);
+      b -= n < 1000;
+      const uint32_t f2 = uint32_t(f0 & mask24) * 100;
+      jeaiii_write_dd(b + 2, f2 >> 24);
+      return b + 4;
+    }
+    const uint64_t f0 = uint64_t(10 * (1ull << 32) / 1e5 + 1) * n;
+    jeaiii_write_fd(b, f0 >> 32);
+    b -= n < 100000;
+    const uint64_t f2 = (f0 & mask32) * 100;
+    jeaiii_write_dd(b + 2, f2 >> 32);
+    const uint64_t f4 = (f2 & mask32) * 100;
+    jeaiii_write_dd(b + 4, f4 >> 32);
+    return b + 6;
+  }
+  const uint64_t f0 = uint64_t(10 * (1ull << 48) / 1e7 + 1) * n >> 16;
+  jeaiii_write_fd(b, f0 >> 32);
+  b -= n < 10000000;
+  const uint64_t f2 = (f0 & mask32) * 100;
+  jeaiii_write_dd(b + 2, f2 >> 32);
+  const uint64_t f4 = (f2 & mask32) * 100;
+  jeaiii_write_dd(b + 4, f4 >> 32);
+  const uint64_t f6 = (f4 & mask32) * 100;
+  jeaiii_write_dd(b + 6, f6 >> 32);
+  return b + 8;
+}
+
+// Caller guarantees z < 10^8. Always writes exactly 8 digits.
+simdjson_really_inline char *jeaiii_8_digits(char *b, uint32_t z) noexcept {
+  constexpr uint64_t mask32 = (uint64_t(1) << 32) - 1;
+  const uint64_t f0 = (uint64_t((1ull << 48) / 1e6 + 1) * z >> 16) + 1;
+  jeaiii_write_dd(b, f0 >> 32);
+  const uint64_t f2 = (f0 & mask32) * 100;
+  jeaiii_write_dd(b + 2, f2 >> 32);
+  const uint64_t f4 = (f2 & mask32) * 100;
+  jeaiii_write_dd(b + 4, f4 >> 32);
+  const uint64_t f6 = (f4 & mask32) * 100;
+  jeaiii_write_dd(b + 6, f6 >> 32);
+  return b + 8;
+}
+
+// Caller guarantees 10^8 <= n < 2^32. Writes 9 or 10 digits.
+simdjson_really_inline char *jeaiii_9_or_10(char *b, uint64_t n) noexcept {
+  constexpr uint64_t mask57 = (uint64_t(1) << 57) - 1;
+  const uint64_t f0 = uint64_t(10 * (1ull << 57) / 1e9 + 1) * n;
+  jeaiii_write_fd(b, f0 >> 57);
+  b -= n < 1000000000;
+  const uint64_t f2 = (f0 & mask57) * 100;
+  jeaiii_write_dd(b + 2, f2 >> 57);
+  const uint64_t f4 = (f2 & mask57) * 100;
+  jeaiii_write_dd(b + 4, f4 >> 57);
+  const uint64_t f6 = (f4 & mask57) * 100;
+  jeaiii_write_dd(b + 6, f6 >> 57);
+  const uint64_t f8 = (f6 & mask57) * 100;
+  jeaiii_write_dd(b + 8, f8 >> 57);
+  return b + 10;
+}
+
+simdjson_really_inline char *write_uint_jeaiii(char *b, uint64_t n) noexcept {
+  if (n < 100000000) {
+    return jeaiii_lt1e8(b, uint32_t(n));
+  }
+  if (n < (uint64_t(1) << 32)) {
+    return jeaiii_9_or_10(b, n);
+  }
+  // At least 10 digits: the low 8 digits, and 2 to 12 digits above them.
+  const uint32_t z = uint32_t(n % 100000000);
+  uint64_t u = n / 100000000;
+  if (u < 100000000) {
+    // u has 2 to 8 digits (if u < 10, n would be below 2^32).
+    b = jeaiii_lt1e8(b, uint32_t(u));
+  } else if (u < (uint64_t(1) << 32)) {
+    b = jeaiii_9_or_10(b, u);
+  } else {
+    // u has 11 or 12 digits: split off 8 more.
+    const uint32_t y = uint32_t(u % 100000000);
+    u /= 100000000;
+    b = jeaiii_lt1e8(b, uint32_t(u)); // 3 or 4 digits
+    b = jeaiii_8_digits(b, y);
   }
-  else {
-    return fast_digit_count_64(static_cast<uint64_t>(v));
-  }
-}
-static const char decimal_table[200] = {
-    0x30, 0x30, 0x30, 0x31, 0x30, 0x32, 0x30, 0x33, 0x30, 0x34, 0x30, 0x35,
-    0x30, 0x36, 0x30, 0x37, 0x30, 0x38, 0x30, 0x39, 0x31, 0x30, 0x31, 0x31,
-    0x31, 0x32, 0x31, 0x33, 0x31, 0x34, 0x31, 0x35, 0x31, 0x36, 0x31, 0x37,
-    0x31, 0x38, 0x31, 0x39, 0x32, 0x30, 0x32, 0x31, 0x32, 0x32, 0x32, 0x33,
-    0x32, 0x34, 0x32, 0x35, 0x32, 0x36, 0x32, 0x37, 0x32, 0x38, 0x32, 0x39,
-    0x33, 0x30, 0x33, 0x31, 0x33, 0x32, 0x33, 0x33, 0x33, 0x34, 0x33, 0x35,
-    0x33, 0x36, 0x33, 0x37, 0x33, 0x38, 0x33, 0x39, 0x34, 0x30, 0x34, 0x31,
-    0x34, 0x32, 0x34, 0x33, 0x34, 0x34, 0x34, 0x35, 0x34, 0x36, 0x34, 0x37,
-    0x34, 0x38, 0x34, 0x39, 0x35, 0x30, 0x35, 0x31, 0x35, 0x32, 0x35, 0x33,
-    0x35, 0x34, 0x35, 0x35, 0x35, 0x36, 0x35, 0x37, 0x35, 0x38, 0x35, 0x39,
-    0x36, 0x30, 0x36, 0x31, 0x36, 0x32, 0x36, 0x33, 0x36, 0x34, 0x36, 0x35,
-    0x36, 0x36, 0x36, 0x37, 0x36, 0x38, 0x36, 0x39, 0x37, 0x30, 0x37, 0x31,
-    0x37, 0x32, 0x37, 0x33, 0x37, 0x34, 0x37, 0x35, 0x37, 0x36, 0x37, 0x37,
-    0x37, 0x38, 0x37, 0x39, 0x38, 0x30, 0x38, 0x31, 0x38, 0x32, 0x38, 0x33,
-    0x38, 0x34, 0x38, 0x35, 0x38, 0x36, 0x38, 0x37, 0x38, 0x38, 0x38, 0x39,
-    0x39, 0x30, 0x39, 0x31, 0x39, 0x32, 0x39, 0x33, 0x39, 0x34, 0x39, 0x35,
-    0x39, 0x36, 0x39, 0x37, 0x39, 0x38, 0x39, 0x39,
-};
+  return jeaiii_8_digits(b, z);
+}
+
+// Writes v at p, which must have to_chars_buffer_size bytes available, and
+// returns the end of what was written.
+simdjson_inline char *write_double(char *p, double v) noexcept {
+#if SIMDJSON_ENABLE_NAN_INF
+  if (simdjson_unlikely(!std::isfinite(v))) {
+    if (std::isnan(v)) {
+      std::memcpy(p, "NaN", 3);
+      return p + 3;
+    }
+    if (v < 0) {
+      *p++ = '-';
+    }
+    std::memcpy(p, "Infinity", 8);
+    return p + 8;
+  }
+#endif
+  return simdjson::internal::to_chars(p, nullptr, v);
+}
 } // namespace internal

 template <typename number_type, typename>
@@ -58509,87 +71747,62 @@ simdjson_inline void string_builder::append(number_type v) noexcept {
     }
   }
   else SIMDJSON_IF_CONSTEXPR(std::is_unsigned<number_type>::value) {
-    // Process 4 digits at a time instead of 2, reducing store operations
-    // and divisions by approximately half for large numbers.
     constexpr size_t max_number_size = 20;
     if (capacity_check(max_number_size)) {
       using unsigned_type = typename std::make_unsigned<number_type>::type;
-      unsigned_type pv = static_cast<unsigned_type>(v);
-      size_t dc = internal::digit_count(pv);
-      char *write_pointer = buffer.get() + position + dc - 1;
-
-      // Process 4 digits per iteration for large numbers
-      while (pv >= 10000) {
-        unsigned_type q = pv / 10000;
-        unsigned_type r = pv % 10000;
-        unsigned_type r_hi = r / 100;  // High 2 digits of remainder
-        unsigned_type r_lo = r % 100;  // Low 2 digits of remainder
-        // Write low 2 digits first (rightmost), then high 2 digits
-        memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
-        memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
-        write_pointer -= 4;
-        pv = q;
-      }
-
-      // Handle remaining 1-4 digits with original 2-digit loop
-      while (pv >= 100) {
-        memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
-        write_pointer -= 2;
-        pv /= 100;
-      }
-      if (pv >= 10) {
-        *write_pointer-- = char('0' + (pv % 10));
-        pv /= 10;
-      }
-      *write_pointer = char('0' + pv);
-      position += dc;
+      char* end = internal::write_uint_jeaiii(
+          buffer.get() + position,
+          static_cast<uint64_t>(static_cast<unsigned_type>(v)));
+      position = end - buffer.get();
     }
   }
   else SIMDJSON_IF_CONSTEXPR(std::is_integral<number_type>::value) {
-    // Same 4-digit batching as unsigned path for signed integers
+    // 19 digits (max abs value of int64_t) + optional minus sign.
     constexpr size_t max_number_size = 20;
     if (capacity_check(max_number_size)) {
       using unsigned_type = typename std::make_unsigned<number_type>::type;
       bool negative = v < 0;
-      unsigned_type pv = static_cast<unsigned_type>(v);
-      if (negative) {
-        pv = 0 - pv; // the 0 is for Microsoft
-      }
-      size_t dc = internal::digit_count(pv);
-      // by always writing the minus sign, we avoid the branch.
+      // 0 - pv (rather than -pv) avoids an MSVC unary-minus warning.
+      unsigned_type pv = negative
+          ? unsigned_type(0) - static_cast<unsigned_type>(v)
+          : static_cast<unsigned_type>(v);
+      // Branchless: always write '-', advance only if negative.
       buffer.get()[position] = '-';
-      position += negative ? 1 : 0;
-      char *write_pointer = buffer.get() + position + dc - 1;
-
-      // Process 4 digits per iteration for large numbers
-      while (pv >= 10000) {
-        unsigned_type q = pv / 10000;
-        unsigned_type r = pv % 10000;
-        unsigned_type r_hi = r / 100;
-        unsigned_type r_lo = r % 100;
-        memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
-        memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
-        write_pointer -= 4;
-        pv = q;
-      }
-
-      // Handle remaining 1-4 digits
-      while (pv >= 100) {
-        memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
-        write_pointer -= 2;
-        pv /= 100;
-      }
-      if (pv >= 10) {
-        *write_pointer-- = char('0' + (pv % 10));
-        pv /= 10;
-      }
-      *write_pointer = char('0' + pv);
-      position += dc;
+      position += negative;
+      char* end = internal::write_uint_jeaiii(
+          buffer.get() + position, static_cast<uint64_t>(pv));
+      position = end - buffer.get();
     }
   }
   else SIMDJSON_IF_CONSTEXPR(std::is_floating_point<number_type>::value) {
-    constexpr size_t max_number_size = 24;
+    // Must reserve to_chars_buffer_size (40): only ~24 chars are emitted,
+    // but to_chars over-writes with fixed-size 16/17-byte copies so the
+    // compiler can inline mem* (see simdjson::internal::to_chars_buffer_size).
+    constexpr size_t max_number_size = simdjson::internal::to_chars_buffer_size;
     if (capacity_check(max_number_size)) {
+#if SIMDJSON_ENABLE_NAN_INF
+      // Check if the input might be NaN or infinity
+      if (simdjson_unlikely(!std::isfinite(v))) {
+        if (std::isnan(v)) {
+          constexpr char nan_literal[] = "NaN";
+          constexpr size_t nan_len = sizeof(nan_literal) - 1;
+
+          std::memcpy(buffer.get() + position, nan_literal, nan_len);
+          position += nan_len;
+        } else {
+          constexpr char inf_literal[] = "Infinity";
+          constexpr size_t inf_len = sizeof(inf_literal) - 1;
+          if (v < 0) {
+            buffer.get()[position] = '-';
+            ++position;
+          }
+          std::memcpy(buffer.get() + position, inf_literal, inf_len);
+          position += inf_len;
+        }
+        return;
+      }
+#endif
+
       // We could specialize for float.
       char *end = simdjson::internal::to_chars(buffer.get() + position, nullptr,
                                                double(v));
@@ -58650,7 +71863,9 @@ simdjson_inline void string_builder::escape_and_append_with_quotes() noexcept {
 #endif

 simdjson_inline void string_builder::append_raw(const char *c) noexcept {
-  size_t len = std::strlen(c);
+  // char_traits::length is constexpr; lets the compiler fold the length
+  // when called with a pointer to a compile-time-constant string.
+  size_t len = std::char_traits<char>::length(c);
   append_raw(c, len);
 }

@@ -58669,6 +71884,14 @@ simdjson_inline void string_builder::append_raw(const char *str,
     position += len;
   }
 }
+
+template <size_t N>
+simdjson_inline void string_builder::append_raw_n(const char *str) noexcept {
+  if (capacity_check(N)) {
+    std::memcpy(buffer.get() + position, str, N);
+    position += N;
+  }
+}
 #if SIMDJSON_SUPPORTS_CONCEPTS
 // Support for optional types (std::optional, etc.)
 template <concepts::optional_type T>
@@ -58698,7 +71921,7 @@ simdjson_inline void string_builder::append(const T &value) {
 #if SIMDJSON_SUPPORTS_RANGES && SIMDJSON_SUPPORTS_CONCEPTS
 // Support for range-based appending (std::ranges::view, etc.)
 template <std::ranges::range R>
-  requires(!std::is_convertible<R, std::string_view>::value && !require_custom_serialization<R>)
+  requires(!std::is_convertible<R, std::string_view>::value && !concepts::optional_type<R> && !require_custom_serialization<R>)
 simdjson_inline void string_builder::append(const R &range) noexcept {
   auto it = std::ranges::begin(range);
   auto end = std::ranges::end(range);
@@ -58993,11 +72216,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
   return __builtin_popcountll(input_num);
 }

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
-                                uint64_t *result) {
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-}

 } // unnamed namespace
 } // namespace rvv_vls
@@ -59724,7 +72942,7 @@ public:
 #if SIMDJSON_SUPPORTS_RANGES && SIMDJSON_SUPPORTS_CONCEPTS
   // Support for range-based appending (std::ranges::view, etc.)
   template <std::ranges::range R>
-requires (!std::is_convertible<R, std::string_view>::value && !require_custom_serialization<R>)
+requires (!std::is_convertible<R, std::string_view>::value && !concepts::optional_type<R> && !require_custom_serialization<R>)
   simdjson_inline void append(const R &range) noexcept;
 #endif
   /**
@@ -59738,6 +72956,15 @@ requires (!std::is_convertible<R, std::string_view>::value && !require_custom_se
    * There is no UTF-8 validation.
    */
   simdjson_inline void append_raw(const char *str, size_t len) noexcept;
+
+  /**
+   * Append exactly N characters from str. The length is a template parameter
+   * so the compiler can fully inline the memcpy with a compile-time-constant
+   * size, avoiding the libc call. Used for compile-time-constant keys in the
+   * reflection struct atom.
+   */
+  template <size_t N>
+  simdjson_inline void append_raw_n(const char *str) noexcept;
 #if SIMDJSON_EXCEPTIONS
   /**
    * Creates an std::string from the written JSON buffer.
@@ -59789,6 +73016,26 @@ requires (!std::is_convertible<R, std::string_view>::value && !require_custom_se
    */
   simdjson_inline size_t size() const noexcept;

+  // ============================================================
+  // Internal hooks for the position-as-local writer in json_builder.h.
+  // These exist so the reflection atom code can hold buffer pointer,
+  // position and capacity in registers across long write chains rather
+  // than reloading them after every char* write (strict aliasing
+  // forces those reloads when accessed via members of *this). User
+  // code should NOT call these directly.
+  // ============================================================
+  simdjson_inline char *unsafe_data() noexcept { return buffer.get(); }
+  simdjson_inline size_t unsafe_position() const noexcept { return position; }
+  simdjson_inline size_t unsafe_capacity() const noexcept { return capacity; }
+  simdjson_inline void unsafe_set_position(size_t p) noexcept { position = p; }
+  /// Make capacity available for at least `n` more bytes after the current
+  /// position. Returns false if the allocation failed.
+  simdjson_inline bool unsafe_grow(size_t needed_total_capacity) noexcept {
+    grow_buffer(needed_total_capacity);
+    return is_valid;
+  }
+  simdjson_inline bool unsafe_is_valid() const noexcept { return is_valid; }
+
 private:
   /**
    * Returns true if we can write at least upcoming_bytes bytes.
@@ -59802,7 +73049,7 @@ private:
    * If the allocation fails, is_valid is set to false. We expect
    * that this function would not be repeatedly called.
    */
-  simdjson_inline void grow_buffer(size_t desired_capacity);
+  inline void grow_buffer(size_t desired_capacity);

   /**
    * We use this helper function to make sure that is_valid is kept consistent.
@@ -59859,6 +73106,7 @@ simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initi
 /* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_STRING_BUILDER_H */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/builder/json_string_builder.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/concepts.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/annotations.h" */
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
 #if SIMDJSON_STATIC_REFLECTION

@@ -59876,64 +73124,370 @@ namespace simdjson {
 namespace rvv_vls {
 namespace builder {

-template <class T>
+// Forward-declare helpers defined in json_string_builder-inl.h so the
+// writer-based atom code below can call them (the -inl.h is not yet
+// included at the point this header is parsed; without these forwards,
+// name lookup falls back to the wrong outer namespace).
+namespace internal {
+simdjson_really_inline char *write_uint_jeaiii(char *p, uint64_t v) noexcept;
+simdjson_inline char *write_double(char *p, double v) noexcept;
+} // namespace internal
+simdjson_really_inline size_t write_string_escaped(const std::string_view input, char *out);
+
+SIMDJSON_PUSH_DISABLE_WARNINGS
+SIMDJSON_DISABLE_GCC_WARNING(-Warray-bounds)
+#if !defined(__clang__)
+SIMDJSON_DISABLE_GCC_WARNING(-Wstringop-overflow)
+#endif
+
+// =============================================================
+// `writer`: position-as-local hot-path writer used by the reflection
+// atom code below. Holds the buffer pointer, write position and
+// capacity in three fields that, once `writer` itself is a stack-local
+// in the caller and all atom() functions are inlined, become true
+// register-resident locals after SROA. Glaze achieves the same effect
+// by passing `B&& b, auto&& ix` through every helper. Holding `pos`
+// in a register (rather than as a member of string_builder) is what
+// breaks the strict-aliasing penalty on every char* write through the
+// buffer, which forces a reload of `b.position` and `b.capacity`
+// after every byte.
+//
+// basic_writer<false> (below) is the unchecked variant: the caller has
+// already reserved enough capacity for everything the write chain can
+// produce (see bound_detail::size_bound), so ensure() compiles away.
+// =============================================================
+template <bool Checked>
+struct basic_writer {
+  static constexpr bool checked = Checked;
+  char *ptr;        // buffer pointer (refreshed after a grow)
+  size_t pos;       // write position (local)
+  size_t cap;       // capacity (refreshed after a grow)
+  string_builder &sb;  // back-ref for grow / sync
+
+  // Snapshot string_builder state into a writer for the duration of
+  // a write chain.
+  simdjson_really_inline basic_writer(string_builder &builder) noexcept
+      : ptr(builder.unsafe_data())
+      , pos(builder.unsafe_position())
+      , cap(builder.unsafe_capacity())
+      , sb(builder) {}
+
+  // Write the local position back to the underlying string_builder.
+  // Caller is responsible for invoking before the writer is dropped
+  // (otherwise data is lost). Idempotent.
+  simdjson_really_inline void sync() noexcept {
+    sb.unsafe_set_position(pos);
+  }
+
+  // Ensure at least `n` more bytes of free capacity. Grows the
+  // underlying buffer if needed (rare path). Returns false on
+  // allocation failure.
+  simdjson_really_inline bool ensure(size_t n) noexcept {
+    // pos <= cap, and cap is the size of a live allocation, so pos + n
+    // cannot wrap when n is a small constant or a compile-time length.
+    // Callers passing a size derived from input (the string atoms) must
+    // bound it against pos themselves. Keep the `pos + n <= cap` form:
+    // `n <= cap - pos` is measurably slower once the serializer is inlined.
+    if (simdjson_likely(pos + n <= cap)) { return true; }
+    return grow_slow(n);
+  }
+
+  simdjson_never_inline bool grow_slow(size_t n) noexcept {
+    // Detect overflow.
+    // This is pedantic except maybe on 32-bit targets.
+    if (simdjson_unlikely(pos + n < pos)) return false;
+    sb.unsafe_set_position(pos);
+    // even if 2*capacity overflows, the (std::max) below will pick the needed value,
+    // so we do not need a separate overflow check here.
+    if (!sb.unsafe_grow((std::max)(cap * 2, pos + n))) {
+      // The string_builder freed its buffer and is now invalid (null buffer,
+      // zero capacity and position). Mirror that state so that every later
+      // ensure() fails too: callers only return from the current atom, and
+      // their callers keep writing.
+      ptr = nullptr;
+      pos = 0;
+      cap = 0;
+      return false;
+    }
+    ptr = sb.unsafe_data();
+    cap = sb.unsafe_capacity();
+    return true;
+  }
+};
+
+// The unchecked writer writes into a raw buffer that the caller sized with
+// serialized_size_bound: it never grows and needs no string_builder.
+template <>
+struct basic_writer<false> {
+  static constexpr bool checked = false;
+  char *ptr;
+  size_t pos;
+
+  simdjson_really_inline basic_writer(char *buffer, size_t position) noexcept
+      : ptr(buffer), pos(position) {}
+
+  simdjson_really_inline bool ensure(size_t) const noexcept { return true; }
+};
+
+using writer = basic_writer<true>;
+using unchecked_writer = basic_writer<false>;
+
+// Bytes reserved past the size bound for an unchecked writer: it may then
+// write a little past the end of what it produces (e.g., copy keys as whole
+// 16-byte blocks).
+inline constexpr size_t unchecked_slack = 64;
+
+consteval size_t padded_key_length(size_t length) {
+  return (length + 15) / 16 * 16;
+}
+
+// === Helper: invoke a string_builder member that writes variable-length
+// content (escape_and_append_with_quotes etc), syncing the writer's local
+// state before the call and reloading after. Used for string fields where
+// rewriting the entire SIMD escape path through the writer would be a much
+// bigger refactor. f may be user code (a with<Adapter> serializer) that
+// throws: the exception then propagates to the caller.
+template <class W, class F>
+simdjson_really_inline void call_through_string_builder(W &w, F &&f) noexcept(noexcept(f(w.sb))) {
+  w.sync();
+  f(w.sb);
+  w.ptr = w.sb.unsafe_data();
+  w.pos = w.sb.unsafe_position();
+  w.cap = w.sb.unsafe_capacity();
+}
+
+// Helpers implementing the serialization side of the annotations (see
+// simdjson/annotations.h) for reflected structures.
+namespace annotation_detail {
+
+// A member is serialized unless it is annotated with skip or skip_serializing.
+consteval bool is_serialized_member(std::meta::info dm) {
+  return !simdjson::detail::has_annotation(dm, ^^simdjson::detail::skip_tag)
+      && !simdjson::detail::has_annotation(dm, ^^simdjson::detail::skip_serializing_tag);
+}
+
+// False when the member has a skip_serializing_if<pred> annotation and
+// pred(value) is true.
+template <auto dm, typename V>
+simdjson_really_inline bool should_serialize(const V &value) {
+  constexpr std::meta::info skip_if_type = simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::skip_serializing_if_t);
+  if constexpr (skip_if_type != std::meta::info{}) {
+    using skip_if = typename [: skip_if_type :];
+    return !skip_if::predicate(value);
+  } else {
+    (void)value;
+    return true;
+  }
+}
+
+// Serialize a member value, through its with<Adapter> annotation when the
+// adapter provides a serialize function.
+template <auto dm, class W, typename V>
+simdjson_really_inline void atom_member(W &w, const V &value) {
+  constexpr std::meta::info with_type = simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::with_t);
+  if constexpr (with_type != std::meta::info{}) {
+    using adapter = typename [: with_type :]::adapter;
+    if constexpr (requires(string_builder &b) { adapter::serialize(b, value); }) {
+      call_through_string_builder(w, [&](string_builder &b) { adapter::serialize(b, value); });
+    } else {
+      atom(w, value);
+    }
+  } else {
+    atom(w, value);
+  }
+}
+
+// Write the "key":value pairs of the members of t (without the braces), each
+// preceded by a comma unless it is the first one. The members of a member
+// annotated with flatten are written in its place.
+template <class W, class T>
+simdjson_really_inline void atom_fields(W &w, const T &t, bool &first) {
+  // Per-field block: ensure key+value worst case, then write key + value
+  // through the writer's local pos. For arithmetic fields, the integer
+  // write happens directly via write_uint_jeaiii on w.ptr+w.pos, so pos
+  // never round-trips through memory.
+  template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+    if constexpr (is_serialized_member(dm)) {
+      if (should_serialize<dm>(t.[:dm:])) {
+        if constexpr (simdjson::detail::has_annotation(dm, ^^simdjson::detail::flatten_tag)) {
+          static_assert(std::meta::is_class_type(simdjson::detail::flattened_type(dm)));
+          using flattened = std::remove_cvref_t<decltype(t.[:dm:])>;
+          static_assert(!concepts::container_but_not_string<flattened> && !concepts::string_view_keyed_map<flattened> &&
+                        !concepts::appendable_containers<flattened> && !concepts::optional_type<flattened> &&
+                        !concepts::smart_pointer<flattened> && !std::is_same_v<flattened, std::string> &&
+                        !std::is_same_v<flattened, std::string_view> && !require_custom_serialization<flattened>,
+                        "simdjson::flatten requires a member whose type is a structure serialized member by member");
+          atom_fields(w, t.[:dm:], first);
+        } else {
+          // Copy the key as whole 16-byte blocks from a zero-padded copy (one
+          // load and one store); ensure() reserves the padded length, and the
+          // unchecked writer has slack past its bound. Prior related work:
+          // jsonifier copies a power-of-two padded key and advances the cursor
+          // by the real length (serialize_impl.hpp, packed_blitter,
+          // https://github.com/nihilai-collective/Jsonifier).
+          constexpr const char* key_name = simdjson::get_json_key_name<dm>();
+          constexpr size_t first_key_len = constevalutil::consteval_to_quoted_escaped(key_name).size() + 1;
+          constexpr size_t rest_key_len = first_key_len + 1;
+          constexpr auto first_key = std::define_static_string(
+              constevalutil::consteval_to_quoted_escaped(key_name) + ":" +
+              std::string(padded_key_length(first_key_len) - first_key_len, '\0'));
+          constexpr auto rest_key = std::define_static_string(
+              std::string(",") + constevalutil::consteval_to_quoted_escaped(key_name) + ":" +
+              std::string(padded_key_length(rest_key_len) - rest_key_len, '\0'));
+          if (!w.ensure(padded_key_length(rest_key_len))) { return; }
+          if (first) {
+            std::memcpy(w.ptr + w.pos, first_key, padded_key_length(first_key_len));
+            w.pos += first_key_len;
+          } else {
+            std::memcpy(w.ptr + w.pos, rest_key, padded_key_length(rest_key_len));
+            w.pos += rest_key_len;
+          }
+          first = false;
+          atom_member<dm>(w, t.[:dm:]);
+        }
+      }
+    }
+  };
+}
+
+} // namespace annotation_detail
+
+template <class W, class T>
   requires(concepts::container_but_not_string<T> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &t) {
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
   auto it = t.begin();
   auto end = t.end();
   if (it == end) {
-    b.append_raw("[]");
+    if (!w.ensure(2)) return;
+    std::memcpy(w.ptr + w.pos, "[]", 2);
+    w.pos += 2;
     return;
   }
-  b.append('[');
-  atom(b, *it);
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = '[';
+  atom(w, *it);
   ++it;
   for (; it != end; ++it) {
-    b.append(',');
-    atom(b, *it);
+    if (!w.ensure(1)) return;
+    w.ptr[w.pos++] = ',';
+    atom(w, *it);
   }
-  b.append(']');
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = ']';
 }

-template <class T>
+template <class W, class T>
   requires(std::is_same_v<T, std::string> ||
            std::is_same_v<T, std::string_view> ||
            std::is_same_v<T, const char *> ||
            std::is_same_v<T, char>)
-constexpr void atom(string_builder &b, const T &t) {
-  b.escape_and_append_with_quotes(t);
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
+  // Inline the escape path through the writer so we never round-trip
+  // pos through memory for string fields (Twitter is dominated by
+  // these -- sync/reload around each string was a real cost).
+  std::string_view input;
+  if constexpr (std::is_same_v<T, char>) {
+    input = std::string_view(&t, 1);
+  } else {
+    input = std::string_view(t);
+  }
+  // Worst-case escape: every byte expands to \uXXXX (6 chars), plus 2 quotes.
+  // Guard against w.pos + 2 + 6 * input.size() wrapping for huge inputs -- if
+  // it wrapped to a small value, ensure() would spuriously succeed and the
+  // subsequent escape would overflow the buffer. max - w.pos cannot wrap, and
+  // size < (max - pos) / 6 implies pos + 6 * size + 6 <= max.
+  // Note that this is pedantic except maybe on 32-bit targets.
+  if constexpr (W::checked) {
+    if (simdjson_unlikely(input.size() >= ((std::numeric_limits<size_t>::max)() - w.pos) / 6)) { return; }
+    if (!w.ensure(2 + 6 * input.size())) { return; }
+  }
+  w.ptr[w.pos++] = '"';
+  w.pos += write_string_escaped(input, w.ptr + w.pos);
+  w.ptr[w.pos++] = '"';
 }

-template <concepts::string_view_keyed_map T>
+template <class W, concepts::string_view_keyed_map T>
   requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &m) {
+simdjson_really_inline constexpr void atom(W &w, const T &m) {
   if (m.empty()) {
-    b.append_raw("{}");
+    if (!w.ensure(2)) return;
+    std::memcpy(w.ptr + w.pos, "{}", 2);
+    w.pos += 2;
     return;
   }
-  b.append('{');
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = '{';
   bool first = true;
   for (const auto& [key, value] : m) {
     if (!first) {
-      b.append(',');
+      if (!w.ensure(1)) return;
+      w.ptr[w.pos++] = ',';
     }
     first = false;
-    // Keys must be convertible to string_view per the concept
-    b.escape_and_append_with_quotes(key);
-    b.append(':');
-    atom(b, value);
+    // Keys must be convertible to string_view per the concept.
+    std::string_view key_sv(key);
+    // Guard against w.pos + 3 + 6 * key_sv.size() wrapping for huge keys, if
+    // it wrapped to a small value, ensure() would spuriously succeed and the
+    // subsequent escape would overflow the buffer. max - w.pos cannot wrap, and
+    // size < (max - pos) / 6 implies pos + 6 * size + 6 <= max.
+    // Note that this is pedantic except maybe on 32-bit targets.
+    if constexpr (W::checked) {
+      if (simdjson_unlikely(key_sv.size() >= ((std::numeric_limits<size_t>::max)() - w.pos) / 6)) { return; }
+      if (!w.ensure(2 + 6 * key_sv.size() + 1)) { return; }
+    }
+    w.ptr[w.pos++] = '"';
+    w.pos += write_string_escaped(key_sv, w.ptr + w.pos);
+    w.ptr[w.pos++] = '"';
+    w.ptr[w.pos++] = ':';
+    atom(w, value);
   }
-  b.append('}');
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = '}';
 }


-template<typename number_type,
+template<class W, typename number_type,
          typename = typename std::enable_if<std::is_arithmetic<number_type>::value && !std::is_same_v<number_type, char>>::type>
-constexpr void atom(string_builder &b, const number_type t) {
-  b.append(t);
+simdjson_really_inline constexpr void atom(W &w, const number_type t) {
+  // Booleans / floats: defer to string_builder (rare path; keeps writer hot
+  // path free of float-formatter machinery). For integers, write directly
+  // via jeaiii using local pos.
+  if constexpr (std::is_same_v<number_type, bool>) {
+    if (t) {
+      if (!w.ensure(4)) return;
+      std::memcpy(w.ptr + w.pos, "true", 4);
+      w.pos += 4;
+    } else {
+      if (!w.ensure(5)) return;
+      std::memcpy(w.ptr + w.pos, "false", 5);
+      w.pos += 5;
+    }
+  } else if constexpr (std::is_floating_point_v<number_type>) {
+    if constexpr (W::checked) {
+      call_through_string_builder(w, [&](string_builder &b) { b.append(t); });
+    } else {
+      w.pos = size_t(internal::write_double(w.ptr + w.pos, double(t)) - w.ptr);
+    }
+  } else if constexpr (std::is_unsigned_v<number_type>) {
+    if (!w.ensure(20)) return;
+    char *end = internal::write_uint_jeaiii(
+        w.ptr + w.pos, static_cast<uint64_t>(t));
+    w.pos = static_cast<size_t>(end - w.ptr);
+  } else {
+    // signed integral
+    if (!w.ensure(20)) return;
+    using U = typename std::make_unsigned<number_type>::type;
+    bool negative = t < 0;
+    U pv = negative ? U(0) - static_cast<U>(t) : static_cast<U>(t);
+    w.ptr[w.pos] = '-';
+    w.pos += negative;
+    char *end = internal::write_uint_jeaiii(
+        w.ptr + w.pos, static_cast<uint64_t>(pv));
+    w.pos = static_cast<size_t>(end - w.ptr);
+  }
 }

-template <class T>
+template <class W, class T>
   requires(std::is_class_v<T> && !concepts::container_but_not_string<T> &&
            !concepts::string_view_keyed_map<T> &&
            !concepts::optional_type<T> &&
@@ -59943,92 +73497,259 @@ template <class T>
            !std::is_same_v<T, std::string_view> &&
            !std::is_same_v<T, const char*> &&
            !std::is_same_v<T, char> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &t) {
-  int i = 0;
-  b.append('{');
-  template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
-    if (i != 0)
-      b.append(',');
-    constexpr auto key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(dm)));
-    b.append_raw(key);
-    b.append(':');
-    atom(b, t.[:dm:]);
-    i++;
-  };
-  b.append('}');
+simdjson_really_inline constexpr void atom(W &w, const T &t) {
+  if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag)) {
+    // A transparent structure is serialized as its single member.
+    constexpr auto dm = simdjson::detail::transparent_member(^^T);
+    annotation_detail::atom_member<dm>(w, t.[:dm:]);
+  } else {
+    bool first = true;
+    if (!w.ensure(1)) { return; }
+    w.ptr[w.pos++] = '{';
+    annotation_detail::atom_fields(w, t, first);
+    if (!w.ensure(1)) { return; }
+    w.ptr[w.pos++] = '}';
+  }
 }

 // Support for optional types (std::optional, etc.)
-template <concepts::optional_type T>
+template <class W, concepts::optional_type T>
   requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &opt) {
+simdjson_really_inline constexpr void atom(W &w, const T &opt) {
   if (opt) {
-    atom(b, opt.value());
+    atom(w, opt.value());
   } else {
-    b.append_raw("null");
+    if (!w.ensure(4)) return;
+    std::memcpy(w.ptr + w.pos, "null", 4);
+    w.pos += 4;
   }
 }

 // Support for smart pointers (std::unique_ptr, std::shared_ptr, etc.)
-template <concepts::smart_pointer T>
+template <class W, concepts::smart_pointer T>
   requires(!require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &ptr) {
+simdjson_really_inline constexpr void atom(W &w, const T &ptr) {
   if (ptr) {
-    atom(b, *ptr);
+    atom(w, *ptr);
   } else {
-    b.append_raw("null");
+    if (!w.ensure(4)) return;
+    std::memcpy(w.ptr + w.pos, "null", 4);
+    w.pos += 4;
   }
 }

 // Support for enums - serialize as string representation using expand approach from P2996R12
-template <typename T>
+template <class W, typename T>
   requires(std::is_enum_v<T> && !require_custom_serialization<T>)
-void atom(string_builder &b, const T &e) {
+simdjson_really_inline void atom(W &w, const T &e) {
 #if SIMDJSON_STATIC_REFLECTION
-  constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
+  static constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
   template for (constexpr auto enum_val : enumerators) {
-    constexpr auto enum_str = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(enum_val)));
+    constexpr auto enum_str = std::define_static_string(constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<enum_val>()));
+    constexpr size_t enum_str_len = std::char_traits<char>::length(enum_str);
     if (e == [:enum_val:]) {
-      b.append_raw(enum_str);
+      if (!w.ensure(enum_str_len)) return;
+      std::memcpy(w.ptr + w.pos, enum_str, enum_str_len);
+      w.pos += enum_str_len;
       return;
     }
   };
   // Fallback to integer if enum value not found
-  atom(b, static_cast<std::underlying_type_t<T>>(e));
+  atom(w, static_cast<std::underlying_type_t<T>>(e));
 #else
   // Fallback: serialize as integer if reflection not available
-  atom(b, static_cast<std::underlying_type_t<T>>(e));
+  atom(w, static_cast<std::underlying_type_t<T>>(e));
 #endif
 }

 // Support for appendable containers that don't have operator[] (sets, etc.)
-template <concepts::appendable_containers T>
+template <class W, concepts::appendable_containers T>
   requires(!concepts::container_but_not_string<T> && !concepts::string_view_keyed_map<T> &&
            !concepts::optional_type<T> && !concepts::smart_pointer<T> &&
            !std::is_same_v<T, std::string> &&
            !std::is_same_v<T, std::string_view> && !std::is_same_v<T, const char*> && !require_custom_serialization<T>)
-constexpr void atom(string_builder &b, const T &container) {
+simdjson_really_inline constexpr void atom(W &w, const T &container) {
   if (container.empty()) {
-    b.append_raw("[]");
+    if (!w.ensure(2)) return;
+    std::memcpy(w.ptr + w.pos, "[]", 2);
+    w.pos += 2;
     return;
   }
-  b.append('[');
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = '[';
   bool first = true;
   for (const auto& item : container) {
     if (!first) {
-      b.append(',');
+      if (!w.ensure(1)) return;
+      w.ptr[w.pos++] = ',';
     }
     first = false;
-    atom(b, item);
+    atom(w, item);
+  }
+  if (!w.ensure(1)) return;
+  w.ptr[w.pos++] = ']';
+}
+
+// =============================================================
+// Size bound: an upper bound on the number of bytes that atom(w, t) writes.
+// Computing it first lets append() reserve the capacity once and then run
+// the whole write chain through an unchecked_writer, without a capacity
+// check before every write. It mirrors the atom() overloads above.
+//
+// Prior related work: jsonifier sizes the document first, counting 6 bytes
+// per string byte, resizes once to that bound plus slack, and writes with
+// no capacity check on each store. to_json() does that resize through
+// std::string::resize_and_overwrite. See serializer.hpp and
+// serialize_impl.hpp in https://github.com/nihilai-collective/Jsonifier
+// and https://nihilai-collective.net/serialization.
+// =============================================================
+namespace bound_detail {
+
+// Whether size_bound covers everything that atom() writes for T: not when a
+// member is serialized by a with<Adapter> serializer, which writes an unknown
+// amount through the string_builder.
+template <class T>
+consteval bool is_bounded() {
+  if constexpr (require_custom_serialization<T>) {
+    return false;
+  } else if constexpr (std::is_same_v<T, std::string> || std::is_same_v<T, std::string_view> ||
+                       std::is_same_v<T, const char *> || std::is_arithmetic_v<T> || std::is_enum_v<T>) {
+    return true;
+  } else if constexpr (concepts::optional_type<T> || concepts::smart_pointer<T>) {
+    return is_bounded<std::remove_cvref_t<decltype(*std::declval<const T &>())>>();
+  } else if constexpr (concepts::string_view_keyed_map<T>) {
+    return is_bounded<std::remove_cvref_t<typename T::mapped_type>>();
+  } else if constexpr (concepts::container_but_not_string<T> || concepts::appendable_containers<T>) {
+    return is_bounded<std::remove_cvref_t<std::ranges::range_value_t<T>>>();
+  } else {
+    bool bounded = true;
+    template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+      if constexpr (annotation_detail::is_serialized_member(dm)) {
+        bounded = bounded && simdjson::detail::annotation_of_template(dm, ^^simdjson::detail::with_t) == std::meta::info{} &&
+                  is_bounded<std::remove_cvref_t<decltype(std::declval<const T &>().[:dm:])>>();
+      }
+    };
+    return bounded;
+  }
+}
+
+template <class T>
+consteval size_t enum_bound() {
+  size_t bound = 20; // the integer fallback
+  template for (constexpr auto enum_val : std::define_static_array(std::meta::enumerators_of(^^T))) {
+    constexpr size_t len = std::char_traits<char>::length(std::define_static_string(
+        constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<enum_val>())));
+    bound = (std::max)(bound, len);
+  };
+  return bound;
+}
+
+template <class T>
+simdjson_really_inline size_t size_bound(const T &t) noexcept;
+
+// Bound for the "key":value pairs of a structure, commas included.
+template <class T>
+simdjson_really_inline size_t fields_bound(const T &t) noexcept {
+  size_t bound = 0;
+  template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
+    if constexpr (annotation_detail::is_serialized_member(dm)) {
+      if constexpr (simdjson::detail::has_annotation(dm, ^^simdjson::detail::flatten_tag)) {
+        bound += fields_bound(t.[:dm:]);
+      } else {
+        constexpr size_t rest_key_len = constevalutil::consteval_to_quoted_escaped(simdjson::get_json_key_name<dm>()).size() + 2;
+        bound += rest_key_len + size_bound(t.[:dm:]);
+      }
+    }
+  };
+  return bound;
+}
+
+template <class T>
+simdjson_really_inline size_t size_bound([[maybe_unused]] const T &t) noexcept {
+  if constexpr (std::is_same_v<T, char>) {
+    return 2 + 6;
+  } else if constexpr (std::is_same_v<T, std::string> || std::is_same_v<T, std::string_view> ||
+                       std::is_same_v<T, const char *>) {
+    // Every byte may become \uXXXX, plus the quotes.
+    return 2 + 6 * std::string_view(t).size();
+  } else if constexpr (std::is_same_v<T, bool>) {
+    return 5;
+  } else if constexpr (std::is_floating_point_v<T>) {
+    return simdjson::internal::to_chars_buffer_size;
+  } else if constexpr (std::is_arithmetic_v<T>) {
+    return 20;
+  } else if constexpr (std::is_enum_v<T>) {
+    return enum_bound<T>();
+  } else if constexpr (concepts::optional_type<T> || concepts::smart_pointer<T>) {
+    return t ? size_bound(*t) : 4;
+  } else if constexpr (concepts::string_view_keyed_map<T>) {
+    size_t bound = 2;
+    for (const auto &[key, value] : t) {
+      // comma, quotes, colon
+      bound += 4 + 6 * std::string_view(key).size() + size_bound(value);
+    }
+    return bound;
+  } else if constexpr (concepts::container_but_not_string<T> || concepts::appendable_containers<T>) {
+    using value_type = std::remove_cvref_t<std::ranges::range_value_t<T>>;
+    if constexpr (std::is_arithmetic_v<value_type> && !std::is_same_v<value_type, char>) {
+      // A fixed bound per element: no need to visit them.
+      return 2 + size_t(std::ranges::distance(t)) * (1 + size_bound(value_type{}));
+    } else {
+      size_t bound = 2;
+#if defined(__GNUC__) && !defined(__clang__)
+#pragma GCC novector // a vector loop is slower on short containers
+#endif
+#pragma GCC unroll 4
+      for (const auto &item : t) {
+        bound += 1 + size_bound(item);
+      }
+      return bound;
+    }
+  } else if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag)) {
+    constexpr auto dm = simdjson::detail::transparent_member(^^T);
+    return size_bound(t.[:dm:]);
+  } else {
+    return 2 + fields_bound(t);
+  }
+}
+
+} // namespace bound_detail
+
+// Write t through an unchecked writer when its size bound is available,
+// reserving that many bytes first, and through the checked writer otherwise.
+template <class T>
+simdjson_really_inline void append_bounded(string_builder &b, const T &t) {
+  // On 32-bit systems, the bound could overflow: keep the checked writer.
+  if constexpr (sizeof(size_t) >= 8 && bound_detail::is_bounded<T>()) {
+    const size_t bound = bound_detail::size_bound(t) + unchecked_slack;
+    const size_t pos = b.unsafe_position();
+    // The bound is a sum of in-memory sizes times a small constant: it cannot
+    // overflow on a 64-bit system. Be pedantic elsewhere.
+    if (sizeof(size_t) >= 8 || bound <= (std::numeric_limits<size_t>::max)() - pos) {
+      const size_t cap = b.unsafe_capacity();
+      // Grow geometrically so that many small appends stay amortized.
+      if (pos + bound <= cap || b.unsafe_grow((std::max)(cap * 2, pos + bound))) {
+        unchecked_writer w(b.unsafe_data(), pos);
+        atom(w, t);
+        b.unsafe_set_position(w.pos);
+      }
+      return;
+    }
   }
-  b.append(']');
+  writer w(b);
+  atom(w, t);
+  w.sync();
 }

-// append functions that delegate to atom functions for primitive types
+// append() -- top-level entry. Each overload constructs a stack-local
+// writer, runs atom(w, t) through the inlined call chain, then syncs
+// the local position back into the string_builder.
 template <class T>
   requires(std::is_arithmetic_v<T> && !std::is_same_v<T, char>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  writer w(b);
+  atom(w, t);
+  w.sync();
 }

 template <class T>
@@ -60036,20 +73757,22 @@ template <class T>
            std::is_same_v<T, std::string_view> ||
            std::is_same_v<T, const char *> ||
            std::is_same_v<T, char>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  writer w(b);
+  atom(w, t);
+  w.sync();
 }

 template <concepts::optional_type T>
   requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  append_bounded(b, t);
 }

 template <concepts::smart_pointer T>
   requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  append_bounded(b, t);
 }

 template <concepts::appendable_containers T>
@@ -60057,14 +73780,14 @@ template <concepts::appendable_containers T>
            !concepts::optional_type<T> && !concepts::smart_pointer<T> &&
            !std::is_same_v<T, std::string> &&
            !std::is_same_v<T, std::string_view> && !std::is_same_v<T, const char*> && !require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  append_bounded(b, t);
 }

 template <concepts::string_view_keyed_map T>
   requires(!require_custom_serialization<T>)
-void append(string_builder &b, const T &t) {
-  atom(b, t);
+simdjson_inline void append(string_builder &b, const T &t) {
+  append_bounded(b, t);
 }

 // works for struct
@@ -60078,39 +73801,15 @@ template <class Z>
            !std::is_same_v<Z, std::string_view> &&
            !std::is_same_v<Z, const char*> &&
            !std::is_same_v<Z, char> && !require_custom_serialization<Z>)
-void append(string_builder &b, const Z &z) {
-  int i = 0;
-  b.append('{');
-  template for (constexpr auto dm : std::define_static_array(std::meta::nonstatic_data_members_of(^^Z, std::meta::access_context::unchecked()))) {
-    if (i != 0)
-      b.append(',');
-    constexpr auto key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(dm)));
-    b.append_raw(key);
-    b.append(':');
-    atom(b, z.[:dm:]);
-    i++;
-  };
-  b.append('}');
+simdjson_inline void append(string_builder &b, const Z &z) {
+  append_bounded(b, z);
 }

 // works for container that have begin() and end() iterators
 template <class Z>
   requires(concepts::container_but_not_string<Z> && !require_custom_serialization<Z>)
-void append(string_builder &b, const Z &z) {
-  auto it = z.begin();
-  auto end = z.end();
-  if (it == end) {
-    b.append_raw("[]");
-    return;
-  }
-  b.append('[');
-  atom(b, *it);
-  ++it;
-  for (; it != end; ++it) {
-    b.append(',');
-    atom(b, *it);
-  }
-  b.append(']');
+simdjson_inline void append(string_builder &b, const Z &z) {
+  append_bounded(b, z);
 }

 template <class Z>
@@ -60121,22 +73820,40 @@ void append(string_builder &b, const Z &z) {


 template <class Z>
-simdjson_warn_unused simdjson_result<std::string> to_json_string(const Z &z, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
-  string_builder b(initial_capacity);
-  append(b, z);
-  std::string_view s;
-  if(auto e = b.view().get(s); e) { return e; }
-  return std::string(s);
+simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
+  if constexpr (sizeof(size_t) >= 8 && bound_detail::is_bounded<Z>()) {
+    // Write straight into s, sized by the bound: no intermediate buffer, no copy.
+    // Prior related work: jsonifier's serializeJson resizes once through
+    // resize_and_overwrite (serializer.hpp).
+    (void)initial_capacity;
+    const size_t bound = bound_detail::size_bound(z) + unchecked_slack;
+    auto write = [&z](char *p) noexcept {
+      unchecked_writer w(p, 0);
+      atom(w, z);
+      return w.pos;
+    };
+#if defined(__cpp_lib_string_resize_and_overwrite) && __cpp_lib_string_resize_and_overwrite >= 202110L
+    s.resize_and_overwrite(bound, [&write](char *p, size_t) noexcept { return write(p); });
+#else
+    s.resize(bound);
+    s.resize(write(s.data()));
+#endif
+    return SUCCESS;
+  } else {
+    string_builder b(initial_capacity);
+    append(b, z);
+    std::string_view view;
+    if(auto e = b.view().get(view); e) { return e; }
+    s.assign(view);
+    return SUCCESS;
+  }
 }

 template <class Z>
-simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
-  string_builder b(initial_capacity);
-  append(b, z);
-  std::string_view view;
-  if(auto e = b.view().get(view); e) { return e; }
-  s.assign(view);
-  return SUCCESS;
+simdjson_warn_unused simdjson_result<std::string> to_json_string(const Z &z, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
+  std::string s;
+  if(auto e = to_json(z, s, initial_capacity); e) { return e; }
+  return s;
 }

 template <class Z>
@@ -60149,40 +73866,41 @@ string_builder& operator<<(string_builder& b, const Z& z) {
 template<constevalutil::fixed_string... FieldNames, typename T>
   requires(std::is_class_v<T> && (sizeof...(FieldNames) > 0))
 void extract_from(string_builder &b, const T &obj) {
-  // Helper to check if a field name matches any of the requested fields
-  auto should_extract = [](std::string_view field_name) constexpr -> bool {
-    return ((FieldNames.view() == field_name) || ...);
-  };
-
-  b.append('{');
+  writer w(b);
+  if (!w.ensure(1)) { w.sync(); return; }
+  w.ptr[w.pos++] = '{';
   bool first = true;
-
   // Iterate through all members of T using reflection
-  template for (constexpr auto mem : std::define_static_array(
-      std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
-
+  static constexpr auto members = std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()));
+  template for (constexpr auto mem : members) {
     if constexpr (std::meta::is_public(mem)) {
-      constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
+      static constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));

       // Only serialize this field if it's in our list of requested fields
-      if constexpr (should_extract(key)) {
-        if (!first) {
-          b.append(',');
+      if constexpr (((FieldNames.view() == key) || ...)) {
+        static constexpr auto first_key = std::define_static_string(
+            constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)) + ":");
+        static constexpr auto rest_key = std::define_static_string(
+            std::string(",") + constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)) + ":");
+        constexpr size_t first_key_len = std::char_traits<char>::length(first_key);
+        constexpr size_t rest_key_len = std::char_traits<char>::length(rest_key);
+        if (!w.ensure(rest_key_len)) { w.sync(); return; }
+        if (first) {
+          std::memcpy(w.ptr + w.pos, first_key, first_key_len);
+          w.pos += first_key_len;
+        } else {
+          std::memcpy(w.ptr + w.pos, rest_key, rest_key_len);
+          w.pos += rest_key_len;
         }
         first = false;
-
-        // Serialize the key
-        constexpr auto quoted_key = std::define_static_string(constevalutil::consteval_to_quoted_escaped(std::meta::identifier_of(mem)));
-        b.append_raw(quoted_key);
-        b.append(':');
-
-        // Serialize the value
-        atom(b, obj.[:mem:]);
+        atom(w, obj.[:mem:]);
       }
     }
   };

-  b.append('}');
+  if (!w.ensure(1)) { w.sync(); return; }
+  w.ptr[w.pos++] = '}';
+  w.sync();
 }

 template<constevalutil::fixed_string... FieldNames, typename T>
@@ -60195,25 +73913,18 @@ simdjson_warn_unused simdjson_result<std::string> extract_from(const T &obj, siz
   return std::string(s);
 }

+SIMDJSON_POP_DISABLE_WARNINGS
+
 } // namespace builder
 } // namespace rvv_vls
 // Alias the function template to 'to' in the global namespace
 template <class Z>
 simdjson_warn_unused simdjson_result<std::string> to_json(const Z &z, size_t initial_capacity = rvv_vls::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
-  rvv_vls::builder::string_builder b(initial_capacity);
-  rvv_vls::builder::append(b, z);
-  std::string_view s;
-  if(auto e = b.view().get(s); e) { return e; }
-  return std::string(s);
+  return rvv_vls::builder::to_json_string(z, initial_capacity);
 }
 template <class Z>
 simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = rvv_vls::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
-  rvv_vls::builder::string_builder b(initial_capacity);
-  rvv_vls::builder::append(b, z);
-  std::string_view view;
-  if(auto e = b.view().get(view); e) { return e; }
-  s.assign(view);
-  return SUCCESS;
+  return rvv_vls::builder::to_json(z, s, initial_capacity);
 }
 // Global namespace function for extract_from
 template<constevalutil::fixed_string... FieldNames, typename T>
@@ -60359,6 +74070,7 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
 /* including simdjson/generic/builder/json_string_builder-inl.h for rvv_vls: #include "simdjson/generic/builder/json_string_builder-inl.h" */
 /* begin file simdjson/generic/builder/json_string_builder-inl.h for rvv_vls */
 #include <array>
+#include <cmath>
 #include <cstring>
 #include <limits>
 #include <type_traits>
@@ -60393,6 +74105,11 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
 #define SIMDJSON_EXPERIMENTAL_HAS_LSX 1
 #endif
 #endif
+#if defined(__loongarch_asx)
+#ifndef SIMDJSON_EXPERIMENTAL_HAS_LASX
+#define SIMDJSON_EXPERIMENTAL_HAS_LASX 1
+#endif
+#endif
 #if defined(__riscv_v_intrinsic) && __riscv_v_intrinsic >= 11000 &&            \
     defined(__riscv_vector)
 #ifndef SIMDJSON_EXPERIMENTAL_HAS_RVV
@@ -60412,6 +74129,9 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
 #endif
 #if SIMDJSON_EXPERIMENTAL_HAS_SSE2
 #include <emmintrin.h>
+#if defined(__AVX2__)
+#include <immintrin.h>
+#endif
 #ifdef _MSC_VER
 #include <intrin.h>
 #endif
@@ -60419,6 +74139,9 @@ simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
 #if SIMDJSON_EXPERIMENTAL_HAS_LSX
 #include <lsxintrin.h>
 #endif
+#if SIMDJSON_EXPERIMENTAL_HAS_LASX
+#include <lasxintrin.h>
+#endif
 #if SIMDJSON_EXPERIMENTAL_HAS_RVV
 #include <riscv_vector.h>
 #endif
@@ -60468,105 +74191,6 @@ inline bool has_json_escapable_byte(uint64_t x) {

 **/

-SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline bool
-simple_needs_escaping(std::string_view v) {
-  for (char c : v) {
-    // a table lookup is faster than a series of comparisons
-    if (json_quotable_character[static_cast<uint8_t>(c)]) {
-      return true;
-    }
-  }
-  return false;
-}
-
-#if SIMDJSON_EXPERIMENTAL_HAS_NEON
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
-  if (view.size() < 16) {
-    return simple_needs_escaping(view);
-  }
-  size_t i = 0;
-  uint8x16_t running = vdupq_n_u8(0);
-  uint8x16_t v34 = vdupq_n_u8(34);
-  uint8x16_t v92 = vdupq_n_u8(92);
-
-  for (; i + 15 < view.size(); i += 16) {
-    uint8x16_t word = vld1q_u8((const uint8_t *)view.data() + i);
-    running = vorrq_u8(running, vceqq_u8(word, v34));
-    running = vorrq_u8(running, vceqq_u8(word, v92));
-    running = vorrq_u8(running, vcltq_u8(word, vdupq_n_u8(32)));
-  }
-  if (i < view.size()) {
-    uint8x16_t word =
-        vld1q_u8((const uint8_t *)view.data() + view.length() - 16);
-    running = vorrq_u8(running, vceqq_u8(word, v34));
-    running = vorrq_u8(running, vceqq_u8(word, v92));
-    running = vorrq_u8(running, vcltq_u8(word, vdupq_n_u8(32)));
-  }
-  return vmaxvq_u32(vreinterpretq_u32_u8(running)) != 0;
-}
-#elif SIMDJSON_EXPERIMENTAL_HAS_SSE2
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
-  if (view.size() < 16) {
-    return simple_needs_escaping(view);
-  }
-  size_t i = 0;
-  __m128i running = _mm_setzero_si128();
-  for (; i + 15 < view.size(); i += 16) {
-
-    __m128i word =
-        _mm_loadu_si128(reinterpret_cast<const __m128i *>(view.data() + i));
-    running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(34)));
-    running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(92)));
-    running = _mm_or_si128(
-        running, _mm_cmpeq_epi8(_mm_subs_epu8(word, _mm_set1_epi8(31)),
-                                _mm_setzero_si128()));
-  }
-  if (i < view.size()) {
-    __m128i word = _mm_loadu_si128(
-        reinterpret_cast<const __m128i *>(view.data() + view.length() - 16));
-    running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(34)));
-    running = _mm_or_si128(running, _mm_cmpeq_epi8(word, _mm_set1_epi8(92)));
-    running = _mm_or_si128(
-        running, _mm_cmpeq_epi8(_mm_subs_epu8(word, _mm_set1_epi8(31)),
-                                _mm_setzero_si128()));
-  }
-  return _mm_movemask_epi8(running) != 0;
-}
-#elif SIMDJSON_EXPERIMENTAL_HAS_PPC64
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
-  if (view.size() < 16) {
-    return simple_needs_escaping(view);
-  }
-  size_t i = 0;
-  __vector unsigned char running = vec_splats((unsigned char)0);
-  __vector unsigned char v34 = vec_splats((unsigned char)34);
-  __vector unsigned char v92 = vec_splats((unsigned char)92);
-  __vector unsigned char v32 = vec_splats((unsigned char)32);
-
-  for (; i + 15 < view.size(); i += 16) {
-    __vector unsigned char word =
-        vec_vsx_ld(0, reinterpret_cast<const unsigned char *>(view.data() + i));
-    running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v34));
-    running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v92));
-    running = vec_or(running,
-        (__vector unsigned char)vec_cmplt(word, v32));
-  }
-  if (i < view.size()) {
-    __vector unsigned char word = vec_vsx_ld(
-        0, reinterpret_cast<const unsigned char *>(view.data() + view.length() - 16));
-    running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v34));
-    running = vec_or(running, (__vector unsigned char)vec_cmpeq(word, v92));
-    running = vec_or(running,
-        (__vector unsigned char)vec_cmplt(word, v32));
-  }
-  return !vec_all_eq(running, vec_splats((unsigned char)0));
-}
-#else
-simdjson_inline bool fast_needs_escaping(std::string_view view) {
-  return simple_needs_escaping(view);
-}
-#endif
-
 // Scalar fallback for finding next quotable character
 SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline size_t
 find_next_json_quotable_character_scalar(const std::string_view view,
@@ -60657,6 +74281,51 @@ find_next_json_quotable_character(const std::string_view view,
   size_t current = len - remaining;
   return find_next_json_quotable_character_scalar(view, current);
 }
+#elif SIMDJSON_EXPERIMENTAL_HAS_LASX
+simdjson_inline size_t
+find_next_json_quotable_character(const std::string_view view,
+                                  size_t location) noexcept {
+  const size_t len = view.size();
+  const uint8_t *ptr =
+      reinterpret_cast<const uint8_t *>(view.data()) + location;
+  size_t remaining = len - location;
+
+  // SIMD constants for characters requiring escape
+  __m256i v34 = __lasx_xvreplgr2vr_b(34);  // '"'
+  __m256i v92 = __lasx_xvreplgr2vr_b(92);  // '\\'
+  __m256i v32 = __lasx_xvreplgr2vr_b(32);  // control char threshold
+
+  while (remaining >= 32) {
+    __m256i word = __lasx_xvld(ptr, 0);
+
+    // Check for quotable characters: '"', '\\', or control chars (< 32)
+    __m256i needs_escape = __lasx_xvseq_b(word, v34);
+    needs_escape = __lasx_xvor_v(needs_escape, __lasx_xvseq_b(word, v92));
+    needs_escape = __lasx_xvor_v(needs_escape, __lasx_xvslt_bu(word, v32));
+
+    if (!__lasx_xbz_v(needs_escape)) {
+      // Found a quotable character - locate it via the four 64-bit lanes
+      uint64_t lane0 = __lasx_xvpickve2gr_du(needs_escape, 0);
+      uint64_t lane1 = __lasx_xvpickve2gr_du(needs_escape, 1);
+      uint64_t lane2 = __lasx_xvpickve2gr_du(needs_escape, 2);
+      uint64_t lane3 = __lasx_xvpickve2gr_du(needs_escape, 3);
+      size_t offset = ptr - reinterpret_cast<const uint8_t *>(view.data());
+      if (lane0 != 0) {
+        return offset + trailing_zeroes(lane0) / 8;
+      } else if (lane1 != 0) {
+        return offset + 8 + trailing_zeroes(lane1) / 8;
+      } else if (lane2 != 0) {
+        return offset + 16 + trailing_zeroes(lane2) / 8;
+      } else {
+        return offset + 24 + trailing_zeroes(lane3) / 8;
+      }
+    }
+    ptr += 32;
+    remaining -= 32;
+  }
+  size_t current = len - remaining;
+  return find_next_json_quotable_character_scalar(view, current);
+}
 #elif SIMDJSON_EXPERIMENTAL_HAS_LSX
 simdjson_inline size_t
 find_next_json_quotable_character(const std::string_view view,
@@ -60812,6 +74481,254 @@ SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline void escape_json_char(char c, char *&o
   }
 }

+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2 || SIMDJSON_EXPERIMENTAL_HAS_NEON
+#define SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE 1
+#endif
+
+#if SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2
+
+using escape_vector = __m128i;
+// Mask bits that each input byte contributes to escape_bitmask().
+static constexpr unsigned escape_mask_bits = 1;
+
+simdjson_inline escape_vector escape_load16(const uint8_t *p) noexcept {
+  return _mm_loadu_si128(reinterpret_cast<const __m128i *>(p));
+}
+
+simdjson_inline void escape_store16(char *out, escape_vector v) noexcept {
+  _mm_storeu_si128(reinterpret_cast<__m128i *>(out), v);
+}
+
+// Builds a vector whose bytes 0..7 come from a and bytes 8..15 from b.
+simdjson_inline escape_vector escape_load8x2(const uint8_t *a,
+                                             const uint8_t *b) noexcept {
+  return _mm_unpacklo_epi64(
+      _mm_loadl_epi64(reinterpret_cast<const __m128i *>(a)),
+      _mm_loadl_epi64(reinterpret_cast<const __m128i *>(b)));
+}
+
+// Builds a vector whose bytes 0..3 come from a and bytes 4..7 from b. The
+// upper half is unspecified; callers only look at the low eight bits.
+simdjson_inline escape_vector escape_load4x2(const uint8_t *a,
+                                             const uint8_t *b) noexcept {
+  int32_t a32, b32;
+  memcpy(&a32, a, 4);
+  memcpy(&b32, b, 4);
+  return _mm_unpacklo_epi32(_mm_cvtsi32_si128(a32), _mm_cvtsi32_si128(b32));
+}
+
+// Sets every bit of byte k when byte k of v is a quotable character ('"',
+// '\\' or a control character).
+simdjson_inline escape_vector escape_flags(escape_vector v) noexcept {
+  const __m128i v34 = _mm_set1_epi8(34); // '"'
+  const __m128i v92 = _mm_set1_epi8(92); // '\\'
+  const __m128i v31 = _mm_set1_epi8(31); // for control char detection
+  __m128i needs_escape = _mm_cmpeq_epi8(v, v34);
+  needs_escape = _mm_or_si128(needs_escape, _mm_cmpeq_epi8(v, v92));
+  return _mm_or_si128(
+      needs_escape, _mm_cmpeq_epi8(_mm_subs_epu8(v, v31), _mm_setzero_si128()));
+}
+
+// True when any byte needs escaping. Kept separate from escape_bitmask
+// because some instruction sets can answer it without leaving the vector
+// register file; on SSE2 the compiler folds the two together.
+simdjson_inline bool escape_any(escape_vector flags) noexcept {
+  return _mm_movemask_epi8(flags) != 0;
+}
+
+// A 16-bit mask with bit k set when byte k needed escaping.
+simdjson_inline uint64_t escape_bitmask(escape_vector flags) noexcept {
+  return uint64_t(uint32_t(_mm_movemask_epi8(flags)));
+}
+
+#else // SIMDJSON_EXPERIMENTAL_HAS_NEON
+
+using escape_vector = uint8x16_t;
+static constexpr unsigned escape_mask_bits = 4;
+
+simdjson_inline escape_vector escape_load16(const uint8_t *p) noexcept {
+  return vld1q_u8(p);
+}
+
+simdjson_inline void escape_store16(char *out, escape_vector v) noexcept {
+  vst1q_u8(reinterpret_cast<uint8_t *>(out), v);
+}
+
+simdjson_inline escape_vector escape_load8x2(const uint8_t *a,
+                                             const uint8_t *b) noexcept {
+  uint64_t a64, b64;
+  memcpy(&a64, a, 8);
+  memcpy(&b64, b, 8);
+  return vreinterpretq_u8_u64(vsetq_lane_u64(b64, vdupq_n_u64(a64), 1));
+}
+
+simdjson_inline escape_vector escape_load4x2(const uint8_t *a,
+                                             const uint8_t *b) noexcept {
+  uint32_t a32, b32;
+  memcpy(&a32, a, 4);
+  memcpy(&b32, b, 4);
+  return vreinterpretq_u8_u32(vsetq_lane_u32(b32, vdupq_n_u32(a32), 1));
+}
+
+simdjson_inline escape_vector escape_flags(escape_vector v) noexcept {
+  uint8x16_t needs_escape = vceqq_u8(v, vdupq_n_u8(34));              // '"'
+  needs_escape = vorrq_u8(needs_escape, vceqq_u8(v, vdupq_n_u8(92))); // '\\'
+  return vorrq_u8(needs_escape, vcltq_u8(v, vdupq_n_u8(32)));
+}
+
+simdjson_inline bool escape_any(escape_vector flags) noexcept {
+  return vmaxvq_u32(vreinterpretq_u32_u8(flags)) != 0;
+}
+
+// Four bits per byte rather than one: escape_block and the tail paths scale
+// their shifts by escape_mask_bits to match.
+simdjson_inline uint64_t escape_bitmask(escape_vector flags) noexcept {
+  uint8x8_t narrowed = vshrn_n_u16(vreinterpretq_u16_u8(flags), 4);
+  return vget_lane_u64(vreinterpret_u64_u8(narrowed), 0);
+}
+
+#endif // instruction set selection
+
+// The escape bitmask of a 16-byte block.
+simdjson_inline uint64_t escape_mask(escape_vector v) noexcept {
+  return escape_bitmask(escape_flags(v));
+}
+
+// Copies n bytes with n < 16, using overlapping loads and stores. It never
+// reads more than n bytes from src, nor writes more than n bytes to dst.
+simdjson_inline void copy_lt16(char *dst, const uint8_t *src,
+                               size_t n) noexcept {
+  if (n >= 8) {
+    memcpy(dst, src, 8);
+    memcpy(dst + n - 8, src + n - 8, 8);
+  } else if (n >= 4) {
+    memcpy(dst, src, 4);
+    memcpy(dst + n - 4, src + n - 4, 4);
+  } else if (n > 0) {
+    dst[0] = char(src[0]);
+    dst[n >> 1] = char(src[n >> 1]);
+    dst[n - 1] = char(src[n - 1]);
+  }
+}
+
+// Escapes the bytes of src in the range [i, blockend), given that m is the
+// (non-zero) escape mask for that range: the escape_mask_bits-wide lane of m
+// at byte k is non-zero when src[i + k] requires escaping. Returns the updated
+// output pointer.
+simdjson_never_inline char *escape_block(const uint8_t *src, char *out,
+                                         size_t i, size_t blockend,
+                                         uint64_t m) noexcept {
+  constexpr uint64_t lane = (uint64_t(1) << escape_mask_bits) - 1;
+
+  size_t pos = i; // first byte not yet copied
+  while (m) {
+    const size_t tz = trailing_zeroes(m);
+    const size_t next = i + tz / escape_mask_bits;
+    // Copy the run of safe bytes that precedes this escape.
+    copy_lt16(out, src + pos, next - pos);
+    out += next - pos;
+    escape_json_char(char(src[next]), out);
+    pos = next + 1;
+    m &= ~(lane << tz);
+  }
+  // Copy whatever follows the last escape.
+  copy_lt16(out, src + pos, blockend - pos);
+  return out + (blockend - pos);
+}
+
+// Writes the escaped version of input to out, returning the number of bytes
+// written.
+simdjson_really_inline size_t write_string_escaped(const std::string_view input, char *out) {
+  const size_t len = input.size();
+  const uint8_t *src = reinterpret_cast<const uint8_t *>(input.data());
+  const char *const initout = out;
+
+  size_t i = 0;
+#if SIMDJSON_EXPERIMENTAL_HAS_SSE2 && defined(__AVX2__)
+  while (i + 32 <= len) {
+    const __m256i word = _mm256_loadu_si256(reinterpret_cast<const __m256i *>(src + i));
+    const __m256i flags = _mm256_or_si256(
+        _mm256_or_si256(_mm256_cmpeq_epi8(word, _mm256_set1_epi8(34)),   // '"'
+                        _mm256_cmpeq_epi8(word, _mm256_set1_epi8(92))),  // '\\'
+        _mm256_cmpeq_epi8(_mm256_subs_epu8(word, _mm256_set1_epi8(31)),
+                          _mm256_setzero_si256()));                      // control
+    const uint32_t mask = uint32_t(_mm256_movemask_epi8(flags));
+    if (simdjson_likely(mask == 0)) {
+      _mm256_storeu_si256(reinterpret_cast<__m256i *>(out), word);
+      out += 32;
+    } else {
+      for (size_t half = 0; half < 32; half += 16) {
+        const uint64_t m = (mask >> half) & 0xFFFF;
+        if (m == 0) {
+          escape_store16(out, escape_load16(src + i + half));
+          out += 16;
+        } else {
+          out = escape_block(src, out, i + half, i + half + 16, m);
+        }
+      }
+    }
+    i += 32;
+  }
+#endif
+  while (i + 16 <= len) {
+    escape_vector word = escape_load16(src + i);
+    escape_vector flags = escape_flags(word);
+    if (simdjson_likely(!escape_any(flags))) {
+      escape_store16(out, word);
+      out += 16;
+    } else {
+      out = escape_block(src, out, i, i + 16, escape_bitmask(flags));
+    }
+    i += 16;
+  }
+  if (i < len) {
+    const size_t rem = len - i;
+    uint64_t m;
+    if (len >= 16) {
+      // The last 16 bytes of the input are in bounds. Bit k of that block's
+      // mask belongs to input position len - 16 + k, so shift it down to align
+      // bit 0 with position i.
+      m = escape_mask(escape_load16(src + len - 16)) >>
+          (escape_mask_bits * (16 - rem));
+    } else if (len >= 8) {
+      // Here i == 0 and rem == len. Two overlapping 8-byte loads cover
+      // [0, 8) and [len - 8, len), which is the whole input since len < 16.
+      uint64_t mm = escape_mask(escape_load8x2(src, src + len - 8));
+      constexpr uint64_t low8 = (uint64_t(1) << (escape_mask_bits * 8)) - 1;
+      m = (mm & low8) |
+          ((mm >> (escape_mask_bits * 8)) << (escape_mask_bits * (len - 8)));
+    } else if (len >= 4) {
+      // Same idea with two overlapping 4-byte loads.
+      uint64_t mm = escape_mask(escape_load4x2(src, src + len - 4));
+      constexpr uint64_t low4 = (uint64_t(1) << (escape_mask_bits * 4)) - 1;
+      m = (mm & low4) | (((mm >> (escape_mask_bits * 4)) & low4)
+                         << (escape_mask_bits * (len - 4)));
+    } else {
+      // Fewer than 4 bytes: at most three table lookups, no need for SIMD.
+      for (size_t k = 0; k < len; k++) {
+        uint8_t c = src[k];
+        if (json_quotable_character[c]) {
+          escape_json_char(char(c), out);
+        } else {
+          *out++ = char(c);
+        }
+      }
+      return size_t(out - initout);
+    }
+    if (m == 0) {
+      copy_lt16(out, src + i, rem);
+      out += rem;
+    } else {
+      out = escape_block(src, out, i, len, m);
+    }
+  }
+  return size_t(out - initout);
+}
+
+#else // SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
 // Writes the escaped version of input to out, returning the number of bytes
 // written. Uses SIMD position finding to locate quotable characters efficiently.
 inline size_t write_string_escaped(const std::string_view input, char *out) {
@@ -60841,9 +74758,13 @@ inline size_t write_string_escaped(const std::string_view input, char *out) {
     escape_json_char(input[location], out);
     location += 1;
   }
-  return out - initout;
+  return size_t(out - initout);
 }

+#endif // SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+#undef SIMDJSON_BUILDER_HAS_BLOCK_ESCAPE
+
+
 simdjson_inline string_builder::string_builder(size_t initial_capacity)
     : buffer(new(std::nothrow) char[initial_capacity]), position(0),
       capacity(buffer.get() != nullptr ? initial_capacity : 0),
@@ -60866,7 +74787,7 @@ simdjson_inline bool string_builder::capacity_check(size_t upcoming_bytes) {
   return is_valid;
 }

-simdjson_inline void string_builder::grow_buffer(size_t desired_capacity) {
+inline void string_builder::grow_buffer(size_t desired_capacity) {
   if (!is_valid) {
     return;
   }
@@ -60922,81 +74843,136 @@ simdjson_inline void string_builder::clear() noexcept {

 namespace internal {

-template <typename number_type, typename = typename std::enable_if<
-                                    std::is_unsigned<number_type>::value>::type>
-simdjson_really_inline int int_log2(number_type x) {
-  return 63 - leading_zeroes(uint64_t(x) | 1);
-}
-
-simdjson_really_inline int fast_digit_count_32(uint32_t x) {
-  static uint64_t table[] = {
-      4294967296,  8589934582,  8589934582,  8589934582,  12884901788,
-      12884901788, 12884901788, 17179868184, 17179868184, 17179868184,
-      21474826480, 21474826480, 21474826480, 21474826480, 25769703776,
-      25769703776, 25769703776, 30063771072, 30063771072, 30063771072,
-      34349738368, 34349738368, 34349738368, 34349738368, 38554705664,
-      38554705664, 38554705664, 41949672960, 41949672960, 41949672960,
-      42949672960, 42949672960};
-  return uint32_t((x + table[int_log2(x)]) >> 32);
-}
-
-simdjson_really_inline int fast_digit_count_64(uint64_t x) {
-  static uint64_t table[] = {9,
-                             99,
-                             999,
-                             9999,
-                             99999,
-                             999999,
-                             9999999,
-                             99999999,
-                             999999999,
-                             9999999999,
-                             99999999999,
-                             999999999999,
-                             9999999999999,
-                             99999999999999,
-                             999999999999999ULL,
-                             9999999999999999ULL,
-                             99999999999999999ULL,
-                             999999999999999999ULL,
-                             9999999999999999999ULL};
-  int y = (19 * int_log2(x) >> 6);
-  y += x > table[y];
-  return y + 1;
-}
-
-template <typename number_type, typename = typename std::enable_if<
-                                    std::is_unsigned<number_type>::value>::type>
-simdjson_really_inline size_t digit_count(number_type v) noexcept {
-  static_assert(sizeof(number_type) == 8 || sizeof(number_type) == 4 ||
-                    sizeof(number_type) == 2 || sizeof(number_type) == 1,
-                "We only support 8-bit, 16-bit, 32-bit and 64-bit numbers");
-  SIMDJSON_IF_CONSTEXPR(sizeof(number_type) <= 4) {
-    return fast_digit_count_32(static_cast<uint32_t>(v));
+// Integer to decimal: James Edward Anhalt III's algorithm
+static const char jeaiii_dd[201] =
+    "00010203040506070809101112131415161718192021222324252627282930313233343536373839"
+    "40414243444546474849505152535455565758596061626364656667686970717273747576777879"
+    "8081828384858687888990919293949596979899";
+static const char jeaiii_fd[201] =
+    "0\0" "1\0" "2\0" "3\0" "4\0" "5\0" "6\0" "7\0" "8\0" "9\0"
+    "10111213141516171819202122232425262728293031323334353637383940414243444546474849"
+    "50515253545556575859606162636465666768697071727374757677787980818283848586878889"
+    "90919293949596979899";
+
+simdjson_really_inline void jeaiii_write_dd(char *p, uint64_t k) noexcept {
+  std::memcpy(p, &jeaiii_dd[2 * k], 2);
+}
+simdjson_really_inline void jeaiii_write_fd(char *p, uint64_t k) noexcept {
+  std::memcpy(p, &jeaiii_fd[2 * k], 2);
+}
+
+// Caller guarantees n < 10^8. Writes 1 to 8 digits.
+simdjson_really_inline char *jeaiii_lt1e8(char *b, uint32_t n) noexcept {
+  constexpr uint64_t mask24 = (uint64_t(1) << 24) - 1;
+  constexpr uint64_t mask32 = (uint64_t(1) << 32) - 1;
+  if (n < 100) {
+    jeaiii_write_fd(b, n);
+    return n < 10 ? b + 1 : b + 2;
+  }
+  if (n < 1000000) {
+    if (n < 10000) {
+      const uint32_t f0 = uint32_t(10 * (1 << 24) / 1e3 + 1) * n;
+      jeaiii_write_fd(b, f0 >> 24);
+      b -= n < 1000;
+      const uint32_t f2 = uint32_t(f0 & mask24) * 100;
+      jeaiii_write_dd(b + 2, f2 >> 24);
+      return b + 4;
+    }
+    const uint64_t f0 = uint64_t(10 * (1ull << 32) / 1e5 + 1) * n;
+    jeaiii_write_fd(b, f0 >> 32);
+    b -= n < 100000;
+    const uint64_t f2 = (f0 & mask32) * 100;
+    jeaiii_write_dd(b + 2, f2 >> 32);
+    const uint64_t f4 = (f2 & mask32) * 100;
+    jeaiii_write_dd(b + 4, f4 >> 32);
+    return b + 6;
+  }
+  const uint64_t f0 = uint64_t(10 * (1ull << 48) / 1e7 + 1) * n >> 16;
+  jeaiii_write_fd(b, f0 >> 32);
+  b -= n < 10000000;
+  const uint64_t f2 = (f0 & mask32) * 100;
+  jeaiii_write_dd(b + 2, f2 >> 32);
+  const uint64_t f4 = (f2 & mask32) * 100;
+  jeaiii_write_dd(b + 4, f4 >> 32);
+  const uint64_t f6 = (f4 & mask32) * 100;
+  jeaiii_write_dd(b + 6, f6 >> 32);
+  return b + 8;
+}
+
+// Caller guarantees z < 10^8. Always writes exactly 8 digits.
+simdjson_really_inline char *jeaiii_8_digits(char *b, uint32_t z) noexcept {
+  constexpr uint64_t mask32 = (uint64_t(1) << 32) - 1;
+  const uint64_t f0 = (uint64_t((1ull << 48) / 1e6 + 1) * z >> 16) + 1;
+  jeaiii_write_dd(b, f0 >> 32);
+  const uint64_t f2 = (f0 & mask32) * 100;
+  jeaiii_write_dd(b + 2, f2 >> 32);
+  const uint64_t f4 = (f2 & mask32) * 100;
+  jeaiii_write_dd(b + 4, f4 >> 32);
+  const uint64_t f6 = (f4 & mask32) * 100;
+  jeaiii_write_dd(b + 6, f6 >> 32);
+  return b + 8;
+}
+
+// Caller guarantees 10^8 <= n < 2^32. Writes 9 or 10 digits.
+simdjson_really_inline char *jeaiii_9_or_10(char *b, uint64_t n) noexcept {
+  constexpr uint64_t mask57 = (uint64_t(1) << 57) - 1;
+  const uint64_t f0 = uint64_t(10 * (1ull << 57) / 1e9 + 1) * n;
+  jeaiii_write_fd(b, f0 >> 57);
+  b -= n < 1000000000;
+  const uint64_t f2 = (f0 & mask57) * 100;
+  jeaiii_write_dd(b + 2, f2 >> 57);
+  const uint64_t f4 = (f2 & mask57) * 100;
+  jeaiii_write_dd(b + 4, f4 >> 57);
+  const uint64_t f6 = (f4 & mask57) * 100;
+  jeaiii_write_dd(b + 6, f6 >> 57);
+  const uint64_t f8 = (f6 & mask57) * 100;
+  jeaiii_write_dd(b + 8, f8 >> 57);
+  return b + 10;
+}
+
+simdjson_really_inline char *write_uint_jeaiii(char *b, uint64_t n) noexcept {
+  if (n < 100000000) {
+    return jeaiii_lt1e8(b, uint32_t(n));
+  }
+  if (n < (uint64_t(1) << 32)) {
+    return jeaiii_9_or_10(b, n);
+  }
+  // At least 10 digits: the low 8 digits, and 2 to 12 digits above them.
+  const uint32_t z = uint32_t(n % 100000000);
+  uint64_t u = n / 100000000;
+  if (u < 100000000) {
+    // u has 2 to 8 digits (if u < 10, n would be below 2^32).
+    b = jeaiii_lt1e8(b, uint32_t(u));
+  } else if (u < (uint64_t(1) << 32)) {
+    b = jeaiii_9_or_10(b, u);
+  } else {
+    // u has 11 or 12 digits: split off 8 more.
+    const uint32_t y = uint32_t(u % 100000000);
+    u /= 100000000;
+    b = jeaiii_lt1e8(b, uint32_t(u)); // 3 or 4 digits
+    b = jeaiii_8_digits(b, y);
   }
-  else {
-    return fast_digit_count_64(static_cast<uint64_t>(v));
-  }
-}
-static const char decimal_table[200] = {
-    0x30, 0x30, 0x30, 0x31, 0x30, 0x32, 0x30, 0x33, 0x30, 0x34, 0x30, 0x35,
-    0x30, 0x36, 0x30, 0x37, 0x30, 0x38, 0x30, 0x39, 0x31, 0x30, 0x31, 0x31,
-    0x31, 0x32, 0x31, 0x33, 0x31, 0x34, 0x31, 0x35, 0x31, 0x36, 0x31, 0x37,
-    0x31, 0x38, 0x31, 0x39, 0x32, 0x30, 0x32, 0x31, 0x32, 0x32, 0x32, 0x33,
-    0x32, 0x34, 0x32, 0x35, 0x32, 0x36, 0x32, 0x37, 0x32, 0x38, 0x32, 0x39,
-    0x33, 0x30, 0x33, 0x31, 0x33, 0x32, 0x33, 0x33, 0x33, 0x34, 0x33, 0x35,
-    0x33, 0x36, 0x33, 0x37, 0x33, 0x38, 0x33, 0x39, 0x34, 0x30, 0x34, 0x31,
-    0x34, 0x32, 0x34, 0x33, 0x34, 0x34, 0x34, 0x35, 0x34, 0x36, 0x34, 0x37,
-    0x34, 0x38, 0x34, 0x39, 0x35, 0x30, 0x35, 0x31, 0x35, 0x32, 0x35, 0x33,
-    0x35, 0x34, 0x35, 0x35, 0x35, 0x36, 0x35, 0x37, 0x35, 0x38, 0x35, 0x39,
-    0x36, 0x30, 0x36, 0x31, 0x36, 0x32, 0x36, 0x33, 0x36, 0x34, 0x36, 0x35,
-    0x36, 0x36, 0x36, 0x37, 0x36, 0x38, 0x36, 0x39, 0x37, 0x30, 0x37, 0x31,
-    0x37, 0x32, 0x37, 0x33, 0x37, 0x34, 0x37, 0x35, 0x37, 0x36, 0x37, 0x37,
-    0x37, 0x38, 0x37, 0x39, 0x38, 0x30, 0x38, 0x31, 0x38, 0x32, 0x38, 0x33,
-    0x38, 0x34, 0x38, 0x35, 0x38, 0x36, 0x38, 0x37, 0x38, 0x38, 0x38, 0x39,
-    0x39, 0x30, 0x39, 0x31, 0x39, 0x32, 0x39, 0x33, 0x39, 0x34, 0x39, 0x35,
-    0x39, 0x36, 0x39, 0x37, 0x39, 0x38, 0x39, 0x39,
-};
+  return jeaiii_8_digits(b, z);
+}
+
+// Writes v at p, which must have to_chars_buffer_size bytes available, and
+// returns the end of what was written.
+simdjson_inline char *write_double(char *p, double v) noexcept {
+#if SIMDJSON_ENABLE_NAN_INF
+  if (simdjson_unlikely(!std::isfinite(v))) {
+    if (std::isnan(v)) {
+      std::memcpy(p, "NaN", 3);
+      return p + 3;
+    }
+    if (v < 0) {
+      *p++ = '-';
+    }
+    std::memcpy(p, "Infinity", 8);
+    return p + 8;
+  }
+#endif
+  return simdjson::internal::to_chars(p, nullptr, v);
+}
 } // namespace internal

 template <typename number_type, typename>
@@ -61024,87 +75000,62 @@ simdjson_inline void string_builder::append(number_type v) noexcept {
     }
   }
   else SIMDJSON_IF_CONSTEXPR(std::is_unsigned<number_type>::value) {
-    // Process 4 digits at a time instead of 2, reducing store operations
-    // and divisions by approximately half for large numbers.
     constexpr size_t max_number_size = 20;
     if (capacity_check(max_number_size)) {
       using unsigned_type = typename std::make_unsigned<number_type>::type;
-      unsigned_type pv = static_cast<unsigned_type>(v);
-      size_t dc = internal::digit_count(pv);
-      char *write_pointer = buffer.get() + position + dc - 1;
-
-      // Process 4 digits per iteration for large numbers
-      while (pv >= 10000) {
-        unsigned_type q = pv / 10000;
-        unsigned_type r = pv % 10000;
-        unsigned_type r_hi = r / 100;  // High 2 digits of remainder
-        unsigned_type r_lo = r % 100;  // Low 2 digits of remainder
-        // Write low 2 digits first (rightmost), then high 2 digits
-        memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
-        memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
-        write_pointer -= 4;
-        pv = q;
-      }
-
-      // Handle remaining 1-4 digits with original 2-digit loop
-      while (pv >= 100) {
-        memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
-        write_pointer -= 2;
-        pv /= 100;
-      }
-      if (pv >= 10) {
-        *write_pointer-- = char('0' + (pv % 10));
-        pv /= 10;
-      }
-      *write_pointer = char('0' + pv);
-      position += dc;
+      char* end = internal::write_uint_jeaiii(
+          buffer.get() + position,
+          static_cast<uint64_t>(static_cast<unsigned_type>(v)));
+      position = end - buffer.get();
     }
   }
   else SIMDJSON_IF_CONSTEXPR(std::is_integral<number_type>::value) {
-    // Same 4-digit batching as unsigned path for signed integers
+    // 19 digits (max abs value of int64_t) + optional minus sign.
     constexpr size_t max_number_size = 20;
     if (capacity_check(max_number_size)) {
       using unsigned_type = typename std::make_unsigned<number_type>::type;
       bool negative = v < 0;
-      unsigned_type pv = static_cast<unsigned_type>(v);
-      if (negative) {
-        pv = 0 - pv; // the 0 is for Microsoft
-      }
-      size_t dc = internal::digit_count(pv);
-      // by always writing the minus sign, we avoid the branch.
+      // 0 - pv (rather than -pv) avoids an MSVC unary-minus warning.
+      unsigned_type pv = negative
+          ? unsigned_type(0) - static_cast<unsigned_type>(v)
+          : static_cast<unsigned_type>(v);
+      // Branchless: always write '-', advance only if negative.
       buffer.get()[position] = '-';
-      position += negative ? 1 : 0;
-      char *write_pointer = buffer.get() + position + dc - 1;
-
-      // Process 4 digits per iteration for large numbers
-      while (pv >= 10000) {
-        unsigned_type q = pv / 10000;
-        unsigned_type r = pv % 10000;
-        unsigned_type r_hi = r / 100;
-        unsigned_type r_lo = r % 100;
-        memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
-        memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
-        write_pointer -= 4;
-        pv = q;
-      }
-
-      // Handle remaining 1-4 digits
-      while (pv >= 100) {
-        memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
-        write_pointer -= 2;
-        pv /= 100;
-      }
-      if (pv >= 10) {
-        *write_pointer-- = char('0' + (pv % 10));
-        pv /= 10;
-      }
-      *write_pointer = char('0' + pv);
-      position += dc;
+      position += negative;
+      char* end = internal::write_uint_jeaiii(
+          buffer.get() + position, static_cast<uint64_t>(pv));
+      position = end - buffer.get();
     }
   }
   else SIMDJSON_IF_CONSTEXPR(std::is_floating_point<number_type>::value) {
-    constexpr size_t max_number_size = 24;
+    // Must reserve to_chars_buffer_size (40): only ~24 chars are emitted,
+    // but to_chars over-writes with fixed-size 16/17-byte copies so the
+    // compiler can inline mem* (see simdjson::internal::to_chars_buffer_size).
+    constexpr size_t max_number_size = simdjson::internal::to_chars_buffer_size;
     if (capacity_check(max_number_size)) {
+#if SIMDJSON_ENABLE_NAN_INF
+      // Check if the input might be NaN or infinity
+      if (simdjson_unlikely(!std::isfinite(v))) {
+        if (std::isnan(v)) {
+          constexpr char nan_literal[] = "NaN";
+          constexpr size_t nan_len = sizeof(nan_literal) - 1;
+
+          std::memcpy(buffer.get() + position, nan_literal, nan_len);
+          position += nan_len;
+        } else {
+          constexpr char inf_literal[] = "Infinity";
+          constexpr size_t inf_len = sizeof(inf_literal) - 1;
+          if (v < 0) {
+            buffer.get()[position] = '-';
+            ++position;
+          }
+          std::memcpy(buffer.get() + position, inf_literal, inf_len);
+          position += inf_len;
+        }
+        return;
+      }
+#endif
+
       // We could specialize for float.
       char *end = simdjson::internal::to_chars(buffer.get() + position, nullptr,
                                                double(v));
@@ -61165,7 +75116,9 @@ simdjson_inline void string_builder::escape_and_append_with_quotes() noexcept {
 #endif

 simdjson_inline void string_builder::append_raw(const char *c) noexcept {
-  size_t len = std::strlen(c);
+  // char_traits::length is constexpr; lets the compiler fold the length
+  // when called with a pointer to a compile-time-constant string.
+  size_t len = std::char_traits<char>::length(c);
   append_raw(c, len);
 }

@@ -61184,6 +75137,14 @@ simdjson_inline void string_builder::append_raw(const char *str,
     position += len;
   }
 }
+
+template <size_t N>
+simdjson_inline void string_builder::append_raw_n(const char *str) noexcept {
+  if (capacity_check(N)) {
+    std::memcpy(buffer.get() + position, str, N);
+    position += N;
+  }
+}
 #if SIMDJSON_SUPPORTS_CONCEPTS
 // Support for optional types (std::optional, etc.)
 template <concepts::optional_type T>
@@ -61213,7 +75174,7 @@ simdjson_inline void string_builder::append(const T &value) {
 #if SIMDJSON_SUPPORTS_RANGES && SIMDJSON_SUPPORTS_CONCEPTS
 // Support for range-based appending (std::ranges::view, etc.)
 template <std::ranges::range R>
-  requires(!std::is_convertible<R, std::string_view>::value && !require_custom_serialization<R>)
+  requires(!std::is_convertible<R, std::string_view>::value && !concepts::optional_type<R> && !require_custom_serialization<R>)
 simdjson_inline void string_builder::append(const R &range) noexcept {
   auto it = std::ranges::begin(range);
   auto end = std::ranges::end(range);
@@ -61448,10 +75409,14 @@ namespace simdjson {
 // Otherwise, amalgamation will fail.
 /* skipped duplicate #include "simdjson/dom/base.h" // for MINIMAL_DOCUMENT_CAPACITY */
 /* skipped duplicate #include "simdjson/implementation.h" */
+/* skipped duplicate #include "simdjson/base.h" */
+/* skipped duplicate #include "simdjson/common_defs.h" */
+/* skipped duplicate #include "simdjson/constevalutil.h" */
 /* skipped duplicate #include "simdjson/padded_string.h" */
 /* skipped duplicate #include "simdjson/padded_string_view.h" */
 /* skipped duplicate #include "simdjson/internal/dom_parser_implementation.h" */
 /* skipped duplicate #include "simdjson/jsonpathutil.h" */
+/* skipped duplicate #include "simdjson/annotations.h" */

 #endif // SIMDJSON_GENERIC_ONDEMAND_DEPENDENCIES_H
 /* end file simdjson/generic/ondemand/dependencies.h */
@@ -61577,7 +75542,14 @@ simdjson_inline int leading_zeroes(uint64_t input_num) {

 /* result might be undefined when input_num is zero */
 simdjson_inline int count_ones(uint64_t input_num) {
+#if SIMDJSON_REGULAR_VISUAL_STUDIO
    return vaddv_u8(vcnt_u8(vcreate_u8(input_num)));
+#else
+   // if the system supports SVE or CSSC, __builtin_popcountll
+   // might be compiled to fewer single instructions. For CSSC,
+   // __builtin_popcountll is compiled to a single instruction.
+   return __builtin_popcountll(input_num);
+#endif// SIMDJSON_REGULAR_VISUAL_STUDIO
 }


@@ -61614,15 +75586,6 @@ simdjson_inline uint64_t zero_leading_bit(uint64_t rev_bits, int leading_zeroes)

 #endif

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2, uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
-  *result = value1 + value2;
-  return *result < value1;
-#else
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-#endif
-}

 } // unnamed namespace
 } // namespace arm64
@@ -61886,6 +75849,7 @@ namespace {
       return vget_lane_u64(
           vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(*this), 4)), 0);
     }
+    // Integer umaxv, not a float compare: flush-to-zero would treat a denormal as zero.
     simdjson_inline bool any() const { return vmaxvq_u32(vreinterpretq_u32_u8(*this)) != 0; }
   };

@@ -62341,7 +76305,7 @@ simdjson_inline escaping escaping::copy_and_find(const uint8_t *src, uint8_t *ds
 /* end file simdjson/arm64/begin.h */
 /* including simdjson/generic/ondemand/amalgamated.h for arm64: #include "simdjson/generic/ondemand/amalgamated.h" */
 /* begin file simdjson/generic/ondemand/amalgamated.h for arm64 */
-#if defined(SIMDJSON_CONDITIONAL_INCLUDE) && !defined(SIMDJSON_GENERIC_BUILDER_DEPENDENCIES_H)
+#if defined(SIMDJSON_CONDITIONAL_INCLUDE) && !defined(SIMDJSON_GENERIC_ONDEMAND_DEPENDENCIES_H)
 #error simdjson/generic/ondemand/dependencies.h must be included before simdjson/generic/ondemand/amalgamated.h!
 #endif

@@ -62390,6 +76354,13 @@ class token_iterator;
 class value;
 class value_iterator;

+#if SIMDJSON_SUPPORTS_RANGES
+class array_range;
+class array_range_iterator;
+class object_range;
+class object_range_iterator;
+#endif // SIMDJSON_SUPPORTS_RANGES
+
 } // namespace ondemand
 } // namespace arm64
 } // namespace simdjson
@@ -62422,6 +76393,9 @@ template <> struct is_builtin_deserializable<arm64::ondemand::object> : std::tru
 template <> struct is_builtin_deserializable<arm64::ondemand::value> : std::true_type {};
 template <> struct is_builtin_deserializable<arm64::ondemand::raw_json_string> : std::true_type {};
 template <> struct is_builtin_deserializable<std::string_view> : std::true_type {};
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template <> struct is_builtin_deserializable<std::u8string_view> : std::true_type {};
+#endif // SIMDJSON_SUPPORTS_CHAR8_T

 template <typename T>
 concept is_builtin_deserializable_v = is_builtin_deserializable<T>::value;
@@ -62439,6 +76413,10 @@ concept nothrow_custom_deserializable = nothrow_tag_invocable<deserialize_tag, V
 template <typename T, typename ValT = arm64::ondemand::value>
 concept nothrow_deserializable = nothrow_custom_deserializable<T, ValT> || is_builtin_deserializable_v<T>;

+// True iff get<T>() on ValT cannot throw: no custom tag_invoke, or that tag_invoke is noexcept.
+template <typename T, typename ValT = arm64::ondemand::value>
+concept nothrow_gettable = !custom_deserializable<T, ValT> || nothrow_custom_deserializable<T, ValT>;
+
 /// Deserialize Tag
 inline constexpr struct deserialize_tag {
   using array_type = arm64::ondemand::array;
@@ -62653,6 +76631,17 @@ public:
    */
   simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> field_key() noexcept;

+  /**
+   * Get the current field's key together with its raw byte length.
+   *
+   * Like field_key(), but also returns the number of raw key bytes (the distance
+   * from the first key byte to the closing quote). The length is recovered from
+   * the structural index -- the next structural token is the ':' -- by stepping
+   * back over any whitespace to the closing quote, avoiding a forward SIMD scan
+   * for the closing quote. Leaves the iterator positioned exactly as field_key().
+   */
+  simdjson_warn_unused simdjson_inline error_code field_key_with_length(raw_json_string &key, std::size_t &len) noexcept;
+
   /**
    * Pass the : in the field and move to its value.
    */
@@ -62805,6 +76794,8 @@ public:
   simdjson_warn_unused simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_warn_unused simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_warn_unused simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<bool> get_bool() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<bool> is_null() noexcept;
   simdjson_warn_unused simdjson_inline bool is_negative() noexcept;
@@ -62823,6 +76814,8 @@ public:
   simdjson_warn_unused simdjson_inline simdjson_result<int64_t> get_root_int64_in_string(bool check_trailing) noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<double> get_root_double(bool check_trailing) noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<double> get_root_double_in_string(bool check_trailing) noexcept;
+  simdjson_warn_unused simdjson_inline simdjson_result<float> get_root_float(bool check_trailing) noexcept;
+  simdjson_warn_unused simdjson_inline simdjson_result<float> get_root_float_in_string(bool check_trailing) noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<bool> get_root_bool(bool check_trailing) noexcept;
   simdjson_warn_unused simdjson_inline bool is_root_negative() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<bool> is_root_integer(bool check_trailing) noexcept;
@@ -62958,6 +76951,15 @@ protected:

   /** @copydoc error_code json_iterator::position() const noexcept; */
   simdjson_inline token_position position() const noexcept;
+  /**
+   * Move the live iterator directly to the given position and depth, without
+   * validating against the parser's per-depth container-start bookkeeping
+   * (unlike json_iterator::reenter_child()). Used to restore a previously
+   * captured mid-container position (see object::revert_position()): that
+   * bookkeeping only tracks each container's own start, not every position
+   * a caller might later capture and revert to, so it does not apply here.
+   */
+  simdjson_inline void reenter_at(token_position position, depth_t depth) noexcept;
   /** @copydoc error_code json_iterator::end_position() const noexcept; */
   simdjson_inline token_position last_position() const noexcept;
   /** @copydoc error_code json_iterator::end_position() const noexcept; */
@@ -63026,9 +77028,12 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+   * Supported types: object, array, raw_json_string, string_view, u8string_view (C++20), uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool
    *
-   * You may use get_double(), get_bool(), get_uint64(), get_int64(),
+   * You may use get_double(), get_float(), get_bool(), get_uint64(), get_int64(),
+   * get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(),
+   * get_float32(), get_float64(),
    * get_object(), get_array(), get_raw_json_string(), or get_string() instead.
    * When SIMDJSON_SUPPORTS_CONCEPTS is set, custom types are also supported.
    *
@@ -63038,7 +77043,7 @@ public:
   template <typename T>
   simdjson_inline simdjson_result<T> get()
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+    noexcept(nothrow_gettable<T, value>)
 #else
     noexcept
 #endif
@@ -63053,7 +77058,8 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+   * Supported types: object, array, raw_json_string, string_view, u8string_view (C++20), uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool
    * If the macro SIMDJSON_SUPPORTS_CONCEPTS is set, then custom types are also supported.
    *
    * @param out This is set to a value of the given type, parsed from the JSON. If there is an error, this may not be initialized.
@@ -63063,7 +77069,7 @@ public:
   template <typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out)
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+    noexcept(nothrow_gettable<T, value>)
 #else
     noexcept
 #endif
@@ -63091,7 +77097,7 @@ public:
       "And you do not seem to have added support for it. Indeed, we have that "
       "simdjson::custom_deserializable<T> is false and the type T is not a default type "
       "such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-      "int64_t, double, or bool.");
+      "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
     static_cast<void>(out); // to get rid of unused errors
     return UNINITIALIZED;
   }
@@ -63100,7 +77106,8 @@ public:
     // immediately fail.
     static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
       "The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-      "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+      "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+      "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
       " get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
       " You may also add support for custom types, see our documentation.");
     static_cast<void>(out); // to get rid of unused errors
@@ -63178,6 +77185,50 @@ public:
    */
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;

+  /**
+   * Cast this JSON value to a 16-bit unsigned integer.
+   *
+   * Calls get_uint64() and checks that the result fits in a uint16_t.
+   *
+   * @returns A 16-bit unsigned integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint16_t.
+   */
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+
+  /**
+   * Cast this JSON value to a 16-bit signed integer.
+   *
+   * Calls get_int64() and checks that the result fits in an int16_t.
+   *
+   * @returns A 16-bit signed integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int16_t.
+   */
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+
+  /**
+   * Cast this JSON value to an 8-bit unsigned integer.
+   *
+   * Calls get_uint64() and checks that the result fits in a uint8_t.
+   *
+   * @returns An 8-bit unsigned integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint8_t.
+   */
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+
+  /**
+   * Cast this JSON value to an 8-bit signed integer.
+   *
+   * Calls get_int64() and checks that the result fits in an int8_t.
+   *
+   * @returns An 8-bit signed integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int8_t.
+   */
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
+
   /**
    * Cast this JSON value to a double.
    *
@@ -63194,6 +77245,53 @@ public:
    */
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;

+  /**
+   * Cast this JSON value to a float (binary32).
+   *
+   * The value is rounded directly to binary32: it is the float nearest to the
+   * JSON number. Note that this may differ from get_double() followed by a cast
+   * to float, which rounds twice.
+   *
+   * @returns A float.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+
+  /**
+   * Cast this JSON value (inside string) to a float (binary32).
+   *
+   * @returns A float.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  /**
+   * Cast this JSON value to a std::float32_t (C++23).
+   *
+   * Same as get_float(): the value is rounded directly to binary32.
+   *
+   * @returns A std::float32_t.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  /**
+   * Cast this JSON value to a std::float64_t (C++23).
+   *
+   * Same as get_double().
+   *
+   * @returns A std::float64_t.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   */
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
+
   /**
    * Cast this JSON value to a string.
    *
@@ -63221,6 +77319,26 @@ public:
    */
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;

+#if SIMDJSON_SUPPORTS_CHAR8_T
+  /**
+   * Cast this JSON value to a C++20 UTF-8 string.
+   *
+   * The string is guaranteed to be valid UTF-8.
+   *
+   * Equivalent to get<std::u8string_view>(). This is a zero-copy alias of
+   * get_string(): the very same bytes are returned, viewed as char8_t.
+   *
+   * Important: a value should be consumed once. Calling get_u8string() twice on the same
+   * value is an error.
+   *
+   * @param allow_replacement Whether to allow a replacement character for unmatched surrogate pairs.
+   * @returns An UTF-8 string. The string is stored in the parser and will be invalidated the next
+   *          time it parses a document or when it is destroyed.
+   * @returns INCORRECT_TYPE if the JSON value is not a string.
+   */
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
   /**
    * Attempts to fill the provided std::string reference with the parsed value of the current string.
    *
@@ -63308,7 +77426,7 @@ public:
   /**
    * Cast this JSON value to an unsigned integer.
    *
-   * @returns A signed 64-bit integer.
+   * @returns A unsigned 64-bit integer.
    * @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit unsigned integer.
    */
   simdjson_inline operator uint64_t() noexcept(false);
@@ -63773,9 +77891,24 @@ public:
   simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template <typename string_type>
   simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -63783,9 +77916,19 @@ public:
   simdjson_inline simdjson_result<bool> get_bool() noexcept;
   simdjson_inline simdjson_result<bool> is_null() noexcept;

-  template<typename T> simdjson_inline simdjson_result<T> get() noexcept;
+  template<typename T> simdjson_inline simdjson_result<T> get()
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, arm64::ondemand::value>);
+#else
+    noexcept;
+#endif

-  template<typename T> simdjson_inline error_code get(T &out) noexcept;
+  template<typename T> simdjson_inline error_code get(T &out)
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, arm64::ondemand::value>);
+#else
+    noexcept;
+#endif

 #if SIMDJSON_EXCEPTIONS
   template <class T>
@@ -64116,6 +78259,7 @@ protected:
   token_position _position{};

   friend class json_iterator;
+  friend class document_stream;
   friend class value_iterator;
   friend class object;
   template <typename... Args>
@@ -64207,6 +78351,9 @@ protected:
    * value of this attribute.
    */
   bool _streaming{false};
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  bool _allow_incomplete_json{false};
+#endif

 public:
   simdjson_inline json_iterator() noexcept = default;
@@ -64231,6 +78378,10 @@ public:
    * start_root_array() and start_root_object().
    */
   simdjson_inline bool streaming() const noexcept;
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  simdjson_inline bool allow_incomplete_json() const noexcept;
+  simdjson_inline size_t remaining_input_length(const uint8_t *json) const noexcept;
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON

   /**
    * Get the root value iterator
@@ -65110,33 +79261,87 @@ public:
    * @param batch_size The batch size to use. MUST be larger than the largest document. The sweet
    *                   spot is cache-related: small enough to fit in cache, yet big enough to
    *                   parse as many documents as possible in one tight loop.
-   *                   Defaults to 10MB, which has been a reasonable sweet spot in our tests.
-   * @param allow_comma_separated (defaults on false) This allows a mode where the documents are
-   *                   separated by commas instead of whitespace. It comes with a performance
-   *                   penalty because the entire document is indexed at once (and the document must be
-   *                   less than 4 GB), and there is no multithreading. In this mode, the batch_size parameter
-   *                   is effectively ignored, as it is set to at least the document size.
+   *                   Defaults to 1MB, which has been a reasonable sweet spot in our tests.
+   * @param allow_comma_separated @deprecated Use stream_format::comma_delimited instead.
+   *                   When true, maps internally to stream_format::comma_delimited.
+   *                   Defaults to false.
    * @return The stream, or an error. An empty input will yield 0 documents rather than an EMPTY error. Errors:
    *         - MEMALLOC if the parser does not have enough capacity and memory allocation fails
    *         - CAPACITY if the parser does not have enough capacity and batch_size > max_capacity.
    *         - other json errors if parsing fails. You should not rely on these errors to always the same for the
    *           same document: they may vary under runtime dispatch (so they may vary depending on your system and hardware).
    */
-  inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
-  inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
-  inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
-  inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size)
+  inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size)
     the string might be automatically padded with up to SIMDJSON_PADDING whitespace characters */
-  inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
-  inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
+  inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @private An rvalue input is destroyed at the end of the full-expression, while the
+   * returned document_stream only holds a pointer to it: iterating the stream would then
+   * read freed memory. These deleted overloads also catch a std::string_view argument,
+   * which would otherwise convert implicitly to a padded_string temporary. */
+  inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size = DEFAULT_BATCH_SIZE) = delete;// unsafe
+  /** @private @overload iterate_many(const std::string &&s, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size = DEFAULT_BATCH_SIZE) = delete;// unsafe
   /** @private We do not want to allow implicit conversion from C string to std::string. */
   simdjson_result<document_stream> iterate_many(const char *buf, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept = delete;

+#ifndef SIMDJSON_DISABLE_DEPRECATED_API
+  /**
+   * @deprecated Use iterate_many with stream_format::comma_delimited instead.
+   */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @private Deleted for the same reason as iterate_many(const std::string &&s, size_t batch_size). */
+  inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size, bool allow_comma_separated) = delete;// unsafe
+  /** @private @overload iterate_many(const std::string &&s, size_t batch_size, bool allow_comma_separated) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size, bool allow_comma_separated) = delete;// unsafe
+#endif // SIMDJSON_DISABLE_DEPRECATED_API
+
+  /**
+   * Parse a stream of JSON documents with explicit format specification.
+   *
+   * @param buf The concatenated JSON documents.
+   * @param len The length of the buffer.
+   * @param batch_size The batch size to use.
+   * @param format The stream format (whitespace_delimited, json_sequence, or comma_delimited).
+   * @return A stream of documents, or an error.
+   */
+  inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(padded_string_view s, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format)
+   *
+   * The string is padded in place if needed (see simdjson::pad), as with iterate_many(std::string &s, size_t batch_size).
+   */
+  inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size, stream_format format) noexcept;
+  /** @private Deleted for the same reason as iterate_many(const std::string &&s, size_t batch_size). */
+  inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+  /** @private @overload iterate_many(const std::string &&s, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+
   /** The capacity of this parser (the largest document it can process). */
   simdjson_pure simdjson_inline size_t capacity() const noexcept;
   /** The maximum capacity of this parser (the largest document it is allowed to process). */
@@ -65264,6 +79469,7 @@ private:
   size_t _capacity{0};
   size_t _max_capacity;
   size_t _max_depth{DEFAULT_MAX_DEPTH};
+  size_t _document_len{0};
   std::unique_ptr<uint8_t[]> string_buf{};

 #if SIMDJSON_DEVELOPMENT_CHECKS
@@ -65326,8 +79532,19 @@ public:
    * Begin array iteration.
    *
    * Part of the std::iterable interface.
+   *
+   * With SIMDJSON_DEVELOPMENT_CHECKS, the iterator locks this array while it is
+   * alive, so that reentrant access (at(), count_elements(), reset(), ...) is
+   * reported as OUT_OF_ORDER_ITERATION.
+   */
+  simdjson_inline simdjson_result<array_iterator> begin() & noexcept;
+  /**
+   * Begin iteration over a temporary array, e.g., `v.get_array().begin()`.
+   *
+   * The iterator does not depend on the array instance and may outlive it, so
+   * it does not lock it.
    */
-  simdjson_inline simdjson_result<array_iterator> begin() noexcept;
+  simdjson_inline simdjson_result<array_iterator> begin() && noexcept;
   /**
    * Sentinel representing the end of the array.
    *
@@ -65458,7 +79675,7 @@ public:
    */
   template <typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out)
-     noexcept(custom_deserializable<T, array> ? nothrow_custom_deserializable<T, array> : true) {
+     noexcept(nothrow_gettable<T, array>) {
     static_assert(custom_deserializable<T, array>);
     return deserialize(*this, out);
   }
@@ -65470,7 +79687,7 @@ public:
    */
   template <typename T>
   simdjson_inline simdjson_result<T> get()
-    noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+    noexcept(nothrow_gettable<T, array>)
   {
     static_assert(std::is_default_constructible<T>::value, "The specified type is not default constructible.");
     T out{};
@@ -65526,6 +79743,10 @@ protected:
    * iter.is_alive() == false indicates iteration is complete.
    */
   value_iterator iter{};
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  bool locked{false};
+  simdjson_inline void set_locked(bool _locked) noexcept;
+#endif

   friend class value;
   friend class document;
@@ -65547,7 +79768,8 @@ public:
   simdjson_inline simdjson_result(error_code error) noexcept; ///< @private
   simdjson_inline simdjson_result() noexcept = default;

-  simdjson_inline simdjson_result<arm64::ondemand::array_iterator> begin() noexcept;
+  simdjson_inline simdjson_result<arm64::ondemand::array_iterator> begin() & noexcept;
+  simdjson_inline simdjson_result<arm64::ondemand::array_iterator> begin() && noexcept;
   simdjson_inline simdjson_result<arm64::ondemand::array_iterator> end() noexcept;
   inline simdjson_result<size_t> count_elements() & noexcept;
   inline simdjson_result<bool> is_empty() & noexcept;
@@ -65567,7 +79789,7 @@ public:
   // TODO: move this code into object-inl.h

   template<typename T>
-  simdjson_inline simdjson_result<T> get() noexcept {
+  simdjson_inline simdjson_result<T> get() noexcept(nothrow_gettable<T, arm64::ondemand::array>) {
     if (error()) { return error(); }
     if constexpr (std::is_same_v<T, arm64::ondemand::array>) {
       return first;
@@ -65575,7 +79797,7 @@ public:
     return first.get<T>();
   }
   template<typename T>
-  simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept {
+  simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept(nothrow_gettable<T, arm64::ondemand::array>) {
     if (error()) { return error(); }
     if constexpr (std::is_same_v<T, arm64::ondemand::array>) {
       out = first;
@@ -65627,6 +79849,15 @@ public:
   /** Create a new, invalid array iterator. */
   simdjson_inline array_iterator() noexcept = default;

+#if SIMDJSON_DEVELOPMENT_CHECKS
+  simdjson_inline ~array_iterator() noexcept;
+
+  simdjson_inline array_iterator(array_iterator&&) noexcept;
+  simdjson_inline array_iterator& operator=(array_iterator&&) noexcept;
+  simdjson_inline array_iterator(const array_iterator&) noexcept;
+  simdjson_inline array_iterator& operator=(const array_iterator&) noexcept;
+#endif
+
   //
   // Iterator interface
   //
@@ -65669,6 +79900,9 @@ public:
 private:
 #if SIMDJSON_DEVELOPMENT_CHECKS
    bool has_been_referenced{false};
+   array* parent{nullptr};
+
+   simdjson_inline array_iterator(const value_iterator &_iter, array* _parent) noexcept;
 #endif
   value_iterator iter{};

@@ -65768,14 +80002,14 @@ public:
   /**
    * Cast this JSON value to an unsigned integer.
    *
-   * @returns A signed 64-bit integer.
+   * @returns A unsigned 64-bit integer.
    * @returns INCORRECT_TYPE If the JSON value is not a 64-bit unsigned integer.
    */
   simdjson_inline simdjson_result<uint64_t> get_uint64() noexcept;
   /**
    * Cast this JSON value (inside string) to an unsigned integer.
    *
-   * @returns A signed 64-bit integer.
+   * @returns A unsigned 64-bit integer.
    * @returns INCORRECT_TYPE If the JSON value is not a 64-bit unsigned integer.
    */
   simdjson_inline simdjson_result<uint64_t> get_uint64_in_string() noexcept;
@@ -65813,6 +80047,46 @@ public:
    * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int32_t.
    */
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  /**
+   * Cast this JSON value to a 16-bit unsigned integer.
+   *
+   * Calls get_uint64() and checks that the result fits in a uint16_t.
+   *
+   * @returns A 16-bit unsigned integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint16_t.
+   */
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  /**
+   * Cast this JSON value to a 16-bit signed integer.
+   *
+   * Calls get_int64() and checks that the result fits in an int16_t.
+   *
+   * @returns A 16-bit signed integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int16_t.
+   */
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  /**
+   * Cast this JSON value to an 8-bit unsigned integer.
+   *
+   * Calls get_uint64() and checks that the result fits in a uint8_t.
+   *
+   * @returns An 8-bit unsigned integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint8_t.
+   */
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  /**
+   * Cast this JSON value to an 8-bit signed integer.
+   *
+   * Calls get_int64() and checks that the result fits in an int8_t.
+   *
+   * @returns An 8-bit signed integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int8_t.
+   */
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   /**
    * Cast this JSON value to a double.
    *
@@ -65828,6 +80102,53 @@ public:
    * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
    */
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+
+  /**
+   * Cast this JSON value to a float (binary32).
+   *
+   * The value is rounded directly to binary32: it is the float nearest to the
+   * JSON number. Note that this may differ from get_double() followed by a cast
+   * to float, which rounds twice.
+   *
+   * @returns A float.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+
+  /**
+   * Cast this JSON value (inside string) to a float (binary32).
+   *
+   * @returns A float.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  /**
+   * Cast this JSON value to a std::float32_t (C++23).
+   *
+   * Same as get_float(): the value is rounded directly to binary32.
+   *
+   * @returns A std::float32_t.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  /**
+   * Cast this JSON value to a std::float64_t (C++23).
+   *
+   * Same as get_double().
+   *
+   * @returns A std::float64_t.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   */
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   /**
    * Cast this JSON value to a string.
    *
@@ -65841,6 +80162,24 @@ public:
    * @returns INCORRECT_TYPE if the JSON value is not a string.
    */
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  /**
+   * Cast this JSON value to a C++20 UTF-8 string.
+   *
+   * The string is guaranteed to be valid UTF-8.
+   *
+   * Equivalent to get<std::u8string_view>(). This is a zero-copy alias of
+   * get_string(): the very same bytes are returned, viewed as char8_t.
+   *
+   * Important: Calling get_u8string() twice on the same document is an error.
+   *
+   * @param allow_replacement Whether to allow a replacement character for unmatched surrogate pairs.
+   * @returns An UTF-8 string. The string is stored in the parser and will be invalidated the next
+   *          time it parses a document or when it is destroyed.
+   * @returns INCORRECT_TYPE if the JSON value is not a string.
+   */
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   /**
    * Attempts to fill the provided std::string reference with the parsed value of the current string.
    *
@@ -65911,9 +80250,12 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool
    *
-   * You may use get_double(), get_bool(), get_uint64(), get_int64(),
+   * You may use get_double(), get_float(), get_bool(), get_uint64(), get_int64(),
+   * get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(),
+   * get_float32(), get_float64(),
    * get_object(), get_array(), get_raw_json_string(), or get_string() instead.
    *
    * @returns A value of the given type, parsed from the JSON.
@@ -65922,7 +80264,7 @@ public:
   template <typename T>
   simdjson_inline simdjson_result<T> get() &
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    noexcept(nothrow_gettable<T, document>)
 #else
     noexcept
 #endif
@@ -65945,7 +80287,7 @@ public:
   template<typename T>
   simdjson_inline simdjson_result<T> get() &&
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    noexcept(nothrow_gettable<T, document>)
 #else
     noexcept
 #endif
@@ -65957,7 +80299,8 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool, value
+   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool, value
    *
    * Be mindful that the document instance must remain in scope while you are accessing object, array and value instances.
    *
@@ -65968,7 +80311,7 @@ public:
   template<typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out) &
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    noexcept(nothrow_gettable<T, document>)
 #else
     noexcept
 #endif
@@ -65981,7 +80324,7 @@ public:
         "And you do not seem to have added support for it. Indeed, we have that "
         "simdjson::custom_deserializable<T> is false and the type T is not a default type "
         "such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-        "int64_t, double, or bool.");
+        "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
       static_cast<void>(out); // to get rid of unused errors
       return UNINITIALIZED;
     }
@@ -65990,7 +80333,8 @@ public:
     // immediately fail.
     static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
       "The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-      "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+      "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+      "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
       " get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
       " You may also add support for custom types, see our documentation.");
     static_cast<void>(out); // to get rid of unused errors
@@ -65999,7 +80343,12 @@ public:
   }

   /** @overload template<typename T> error_code get(T &out) & noexcept */
-  template<typename T> simdjson_deprecated simdjson_inline error_code get(T &out) && noexcept;
+  template<typename T> simdjson_deprecated simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, document>);
+#else
+    noexcept;
+#endif

 #if SIMDJSON_EXCEPTIONS
   /**
@@ -66033,24 +80382,24 @@ public:
   /**
    * Cast this JSON value to an unsigned integer.
    *
-   * @returns A signed 64-bit integer.
+   * @returns A unsigned 64-bit integer.
    * @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit unsigned integer.
    */
-  simdjson_inline operator uint64_t() noexcept(false);
+  explicit simdjson_inline operator uint64_t() noexcept(false);
   /**
    * Cast this JSON value to a signed integer.
    *
    * @returns A signed 64-bit integer.
    * @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit integer.
    */
-  simdjson_inline operator int64_t() noexcept(false);
+  explicit simdjson_inline operator int64_t() noexcept(false);
   /**
    * Cast this JSON value to a double.
    *
    * @returns A double.
    * @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a valid floating-point number.
    */
-  simdjson_inline operator double() noexcept(false);
+  explicit simdjson_inline operator double() noexcept(false);
   /**
    * Cast this JSON value to a string.
    *
@@ -66060,7 +80409,7 @@ public:
    *          time it parses a document or when it is destroyed.
    * @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not a string.
    */
-  simdjson_inline operator std::string_view() noexcept(false) simdjson_lifetime_bound;
+  explicit simdjson_inline operator std::string_view() noexcept(false) simdjson_lifetime_bound;
   /**
    * Cast this JSON value to a raw_json_string.
    *
@@ -66069,14 +80418,14 @@ public:
    * @returns A pointer to the raw JSON for the given string.
    * @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not a string.
    */
-  simdjson_inline operator raw_json_string() noexcept(false) simdjson_lifetime_bound;
+  explicit simdjson_inline operator raw_json_string() noexcept(false) simdjson_lifetime_bound;
   /**
    * Cast this JSON value to a bool.
    *
    * @returns A bool value.
    * @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not true or false.
    */
-  simdjson_inline operator bool() noexcept(false);
+  explicit simdjson_inline operator bool() noexcept(false);
   /**
    * Cast this JSON value to a value when the document is an object or an array.
    *
@@ -66571,9 +80920,24 @@ public:
   simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template <typename string_type>
   simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -66585,7 +80949,7 @@ public:
   template <typename T>
   simdjson_inline simdjson_result<T> get() &
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    noexcept(nothrow_gettable<T, document_reference>)
 #else
     noexcept
 #endif
@@ -66598,7 +80962,8 @@ public:
   template<typename T>
   simdjson_inline simdjson_result<T> get() &&
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    // Forwards to document::get<T>(), so the document customization decides.
+    noexcept(nothrow_gettable<T, document>)
 #else
     noexcept
 #endif
@@ -66610,7 +80975,8 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool, value
+   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool, value
    *
    * Be mindful that the document instance must remain in scope while you are accessing object, array and value instances.
    *
@@ -66621,7 +80987,7 @@ public:
   template<typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out) &
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document_reference> : true)
+    noexcept(nothrow_gettable<T, document_reference>)
 #else
     noexcept
 #endif
@@ -66634,7 +81000,7 @@ public:
         "And you do not seem to have added support for it. Indeed, we have that "
         "simdjson::custom_deserializable<T> is false and the type T is not a default type "
         "such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-        "int64_t, double, or bool.");
+        "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
       static_cast<void>(out); // to get rid of unused errors
       return UNINITIALIZED;
     }
@@ -66643,7 +81009,8 @@ public:
     // immediately fail.
     static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
       "The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-      "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+      "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+      "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
       " get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
       " You may also add support for custom types, see our documentation.");
     static_cast<void>(out); // to get rid of unused errors
@@ -66652,7 +81019,12 @@ public:
   }

   /** @overload template<typename T> error_code get(T &out) & noexcept */
-  template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+  template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, document_reference>);
+#else
+    noexcept;
+#endif
   simdjson_inline simdjson_result<std::string_view> raw_json() noexcept;
 #if SIMDJSON_STATIC_REFLECTION
   template<constevalutil::fixed_string... FieldNames, typename T>
@@ -66665,12 +81037,12 @@ public:
   explicit simdjson_inline operator T() noexcept(false);
   simdjson_inline operator array() & noexcept(false);
   simdjson_inline operator object() & noexcept(false);
-  simdjson_inline operator uint64_t() noexcept(false);
-  simdjson_inline operator int64_t() noexcept(false);
-  simdjson_inline operator double() noexcept(false);
-  simdjson_inline operator std::string_view() noexcept(false);
-  simdjson_inline operator raw_json_string() noexcept(false);
-  simdjson_inline operator bool() noexcept(false);
+  explicit simdjson_inline operator uint64_t() noexcept(false);
+  explicit simdjson_inline operator int64_t() noexcept(false);
+  explicit simdjson_inline operator double() noexcept(false);
+  explicit simdjson_inline operator std::string_view() noexcept(false);
+  explicit simdjson_inline operator raw_json_string() noexcept(false);
+  explicit simdjson_inline operator bool() noexcept(false);
   simdjson_inline operator value() noexcept(false);
 #endif
   simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -66732,9 +81104,24 @@ public:
   simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template <typename string_type>
   simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -66743,11 +81130,31 @@ public:
   simdjson_inline simdjson_result<arm64::ondemand::value> get_value() noexcept;
   simdjson_inline simdjson_result<bool> is_null() noexcept;

-  template<typename T> simdjson_inline simdjson_result<T> get() & noexcept;
-  template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() && noexcept;
+  template<typename T> simdjson_inline simdjson_result<T> get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, arm64::ondemand::document>);
+#else
+    noexcept;
+#endif
+  template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, arm64::ondemand::document>);
+#else
+    noexcept;
+#endif

-  template<typename T> simdjson_inline error_code get(T &out) & noexcept;
-  template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+  template<typename T> simdjson_inline error_code get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, arm64::ondemand::document>);
+#else
+    noexcept;
+#endif
+  template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, arm64::ondemand::document>);
+#else
+    noexcept;
+#endif
 #if SIMDJSON_EXCEPTIONS

   using arm64::implementation_simdjson_result_base<arm64::ondemand::document>::operator*;
@@ -66756,12 +81163,12 @@ public:
   explicit simdjson_inline operator T() noexcept(false);
   simdjson_inline operator arm64::ondemand::array() & noexcept(false);
   simdjson_inline operator arm64::ondemand::object() & noexcept(false);
-  simdjson_inline operator uint64_t() noexcept(false);
-  simdjson_inline operator int64_t() noexcept(false);
-  simdjson_inline operator double() noexcept(false);
-  simdjson_inline operator std::string_view() noexcept(false);
-  simdjson_inline operator arm64::ondemand::raw_json_string() noexcept(false);
-  simdjson_inline operator bool() noexcept(false);
+  explicit simdjson_inline operator uint64_t() noexcept(false);
+  explicit simdjson_inline operator int64_t() noexcept(false);
+  explicit simdjson_inline operator double() noexcept(false);
+  explicit simdjson_inline operator std::string_view() noexcept(false);
+  explicit simdjson_inline operator arm64::ondemand::raw_json_string() noexcept(false);
+  explicit simdjson_inline operator bool() noexcept(false);
   simdjson_inline operator arm64::ondemand::value() noexcept(false);
 #endif
   simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -66827,9 +81234,24 @@ public:
   simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template <typename string_type>
   simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -66838,22 +81260,42 @@ public:
   simdjson_inline simdjson_result<arm64::ondemand::value> get_value() noexcept;
   simdjson_inline simdjson_result<bool> is_null() noexcept;

-  template<typename T> simdjson_inline simdjson_result<T> get() & noexcept;
-  template<typename T> simdjson_inline simdjson_result<T> get() && noexcept;
+  template<typename T> simdjson_inline simdjson_result<T> get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, arm64::ondemand::document_reference>);
+#else
+    noexcept;
+#endif
+  template<typename T> simdjson_inline simdjson_result<T> get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, arm64::ondemand::document_reference>);
+#else
+    noexcept;
+#endif

-  template<typename T> simdjson_inline error_code get(T &out) & noexcept;
-  template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+  template<typename T> simdjson_inline error_code get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, arm64::ondemand::document_reference>);
+#else
+    noexcept;
+#endif
+  template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, arm64::ondemand::document_reference>);
+#else
+    noexcept;
+#endif
 #if SIMDJSON_EXCEPTIONS
   template <class T>
   explicit simdjson_inline operator T() noexcept(false);
   simdjson_inline operator arm64::ondemand::array() & noexcept(false);
   simdjson_inline operator arm64::ondemand::object() & noexcept(false);
-  simdjson_inline operator uint64_t() noexcept(false);
-  simdjson_inline operator int64_t() noexcept(false);
-  simdjson_inline operator double() noexcept(false);
-  simdjson_inline operator std::string_view() noexcept(false);
-  simdjson_inline operator arm64::ondemand::raw_json_string() noexcept(false);
-  simdjson_inline operator bool() noexcept(false);
+  explicit simdjson_inline operator uint64_t() noexcept(false);
+  explicit simdjson_inline operator int64_t() noexcept(false);
+  explicit simdjson_inline operator double() noexcept(false);
+  explicit simdjson_inline operator std::string_view() noexcept(false);
+  explicit simdjson_inline operator arm64::ondemand::raw_json_string() noexcept(false);
+  explicit simdjson_inline operator bool() noexcept(false);
   simdjson_inline operator arm64::ondemand::value() noexcept(false);
 #endif
   simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -67021,10 +81463,7 @@ public:
    *   }
    *   size_t truncated = stream.truncated_bytes();
    *
-   * IMPORTANT: this value is only meaningful under the conditions below. It is
-   * computed from stage-1 bookkeeping, and outside these conditions it is not
-   * merely imprecise, it is arbitrary -- it can exceed size_in_bytes() or wrap
-   * around to a huge value. Check it only when both of the following hold:
+   * IMPORTANT: this value is only meaningful under the conditions below.
    *
    *   - you iterated all the way to the end of the stream;
    *   - no document reported an error. Iteration stops at the first failed
@@ -67033,6 +81472,9 @@ public:
    * If you need to know about a truncated tail outside those conditions, track
    * it yourself from the last successful document (see iterator::current_index()
    * and iterator::source()).
+   *
+   * An empty input (zero bytes) or an input made only of white space contains
+   * no document: truncated_bytes() returns zero.
    */
   inline size_t truncated_bytes() const noexcept;

@@ -67092,7 +81534,10 @@ public:
      *
      * The returned string_view instance is simply a map to the (unparsed)
      * source string: it may thus include white-space characters and all manner
-     * of padding.
+     * of padding. It spans the whole current document, whether or not you
+     * have already accessed (part of) the document. Thus
+     * current_index() + source().size() is the offset just past the end of the
+     * current document, which is useful when reading a stream in chunks.
      *
      * This function (source()) is experimental and the usage
      * may change in future versions of simdjson: we find the API somewhat
@@ -67146,13 +81591,16 @@ private:
    * @param buf is the raw byte buffer we need to process
    * @param len is the length of the raw byte buffer in bytes
    * @param batch_size is the size of the windows (must be strictly greater or equal to the largest JSON document)
+   * @param allow_comma_separated whether to allow comma-separated documents
+   * @param format the stream format
    */
   simdjson_inline document_stream(
     ondemand::parser &parser,
     const uint8_t *buf,
     size_t len,
     size_t batch_size,
-    bool allow_comma_separated
+    bool allow_comma_separated,
+    stream_format format = stream_format::whitespace_delimited
   ) noexcept;

   /**
@@ -67186,8 +81634,23 @@ private:
    */
   inline void next() noexcept;

-  /** Move the json_iterator of the document to the location of the next document in the stream. */
+  /**
+   * Move the json_iterator of the document to the location of the next document
+   * in the stream.
+   *
+   * For formats with a document delimiter (`newline_delimited`, `json_sequence`),
+   * when the iterator is still inside the current document (`depth() > 0`), this
+   * may jump to the next delimiter instead of walking remaining structurals. That
+   * jump does not structure-validate the unread remainder.
+   */
   inline void next_document() noexcept;
+  /** Byte that ends a document under `format`, or 0 if there is none. */
+  simdjson_inline uint8_t document_delimiter() const noexcept;
+  /**
+   * Position the iterator at the first structural at or past the next
+   * `delimiter` in the current batch. Returns false if none is found.
+   */
+  simdjson_inline bool skip_to_delimiter(uint8_t delimiter) noexcept;

   /** Get the next document index. */
   inline size_t next_batch_start() const noexcept;
@@ -67201,6 +81664,7 @@ private:
   size_t len;
   size_t batch_size;
   bool allow_comma_separated;
+  stream_format format;
   /**
    * We are going to use just one document instance. The document owns
    * the json_iterator. It implies that we only ever pass a reference
@@ -67227,7 +81691,7 @@ private:
   /** The error returned from the stage 1 thread. */
   error_code stage1_thread_error{UNINITIALIZED};
   /** The thread used to run stage 1 against the next batch in the background. */
-  std::unique_ptr<stage1_worker> worker{new(std::nothrow) stage1_worker()};
+  std::unique_ptr<stage1_worker> worker{};
   /**
    * The parser used to run stage 1 in the background. Will be swapped
    * with the regular parser when finished.
@@ -67302,6 +81766,16 @@ public:
    * call it again nor can you call key().
    */
   simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  /**
+   * Get the key as a C++20 u8string_view. This is a zero-copy alias of
+   * unescaped_key(): the very same bytes are returned, viewed as char8_t.
+   *
+   * This consumes the key: once you have called unescaped_u8key(), you cannot
+   * call it again nor can you call key().
+   */
+  simdjson_inline simdjson_warn_unused simdjson_result<std::u8string_view> unescaped_u8key(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   /**
    * Get the key as a string_view (for higher speed, consider raw_key).
    * We deliberately use a more cumbersome name (unescaped_key) to force users
@@ -67334,6 +81808,16 @@ public:
    * you can safely call it repeatedly. Use unescaped_key() to get the unescaped key.
    */
   simdjson_inline std::string_view escaped_key() const noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  /**
+   * Get the key as a C++20 u8string_view. This is a zero-copy alias of
+   * escaped_key(): the very same bytes are returned, viewed as char8_t.
+   * The string is unprocessed, so it may contain escape characters
+   * (e.g., \uXXXX or \n). It does not count as a consumption of the content:
+   * you can safely call it repeatedly.
+   */
+  simdjson_inline std::u8string_view escaped_u8key() const noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   /**
    * Get the field value.
    */
@@ -67365,11 +81849,17 @@ public:
   simdjson_inline simdjson_result() noexcept = default;

   simdjson_inline simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> unescaped_u8key(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template<typename string_type>
   simdjson_inline error_code unescaped_key(string_type &receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<arm64::ondemand::raw_json_string> key() noexcept;
   simdjson_inline simdjson_result<std::string_view> key_raw_json_token() noexcept;
   simdjson_inline simdjson_result<std::string_view> escaped_key() noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> escaped_u8key() noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   simdjson_inline simdjson_result<arm64::ondemand::value> value() noexcept;
 };

@@ -67377,6 +81867,1398 @@ public:

 #endif // SIMDJSON_GENERIC_ONDEMAND_FIELD_H
 /* end file simdjson/generic/ondemand/field.h for arm64 */
+/* including simdjson/generic/ondemand/key_selector.h for arm64: #include "simdjson/generic/ondemand/key_selector.h" */
+/* begin file simdjson/generic/ondemand/key_selector.h for arm64 */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+#define SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #include "simdjson/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/common_defs.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/constevalutil.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/raw_json_string.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#include <array>
+#include <string>      // std::string (key_selector::describe)
+#include <string_view>
+#include <cstddef>
+#include <cstdint>
+#include <cstring>     // std::memcpy (portable unaligned window load)
+#include <utility>     // std::index_sequence (window candidate dispatch)
+#include <type_traits> // std::integral_constant (window candidate dispatch)
+
+// AArch64 only: 32-bit ARM (ARMv7) defines __ARM_NEON too, but lacks the
+// A64-only horizontal reductions (vmaxvq_u32) used below. See issue #2885.
+#if defined(__aarch64__)
+  #include <arm_neon.h>
+  #define SIMDJSON_KEY_SELECTOR_HAS_NEON 1
+#else
+  #define SIMDJSON_KEY_SELECTOR_HAS_NEON 0
+#endif
+#if defined(__SSE2__)
+  #include <emmintrin.h>
+  #define SIMDJSON_KEY_SELECTOR_HAS_SSE2 1
+#else
+  #define SIMDJSON_KEY_SELECTOR_HAS_SSE2 0
+#endif
+#if defined(__loongarch_sx)
+  #include <lsxintrin.h>
+  #define SIMDJSON_KEY_SELECTOR_HAS_LSX 1
+#else
+  #define SIMDJSON_KEY_SELECTOR_HAS_LSX 0
+#endif
+
+#if SIMDJSON_SUPPORTS_CONCEPTS
+
+namespace simdjson {
+namespace arm64 {
+namespace ondemand {
+
+namespace key_selector_detail {
+
+// Since not constexpr, triggers compile-time error.
+inline void compile_time_error(const char* message) noexcept { (void)message; }
+
+// ============================================================================
+// Compile-time perfect-hash generator.
+// It scales to ~100 keys at compile time by determining association values one (position, character)
+// symbol at a time (gperf-style) instead of an exhaustive offset search, and
+// falls back to a Hash-and-Displace construction for large/awkward key sets.
+//
+// Only flat tables survive to runtime; the lookup is a few additions plus a
+// single SIMD key comparison (see match_raw below).
+// ============================================================================
+
+// Maximum number of character positions the gperf hash may combine.
+static constexpr std::size_t MAX_POSITIONS = 16;
+// Sentinel "position" meaning "the last character of the key".
+static constexpr std::size_t LAST_CHAR = std::size_t(-1);
+// Runtime-encoded sentinels (stored in uint8 tables).
+static constexpr std::uint8_t POS_LAST_CHAR = 0xFF; // positions_[i] == last char
+static constexpr std::uint8_t HD_MODE       = 0xFF; // num_positions == H&D mode
+// Flags stored in positions[2] in H&D mode to select the key-hash variant.
+static constexpr std::size_t HD_HASH_2BYTE_FLAG = 2;
+static constexpr std::size_t HD_HASH_4BYTE_FLAG = 4;
+
+constexpr std::size_t next_power_of_2(std::size_t n) noexcept {
+    if (n == 0) { return 1; }
+    std::size_t p = 1;
+    while (p < n) { p <<= 1; }
+    return p;
+}
+
+// Character at a given position (LAST_CHAR means last character), or 256 if out
+// of bounds.
+constexpr std::size_t char_at(std::string_view key, std::size_t pos) noexcept {
+    if (pos == LAST_CHAR) {
+        if (key.empty()) { return 256; }
+        return static_cast<unsigned char>(key[key.size() - 1]);
+    }
+    if (pos >= key.size()) { return 256; }
+    return static_cast<unsigned char>(key[pos]);
+}
+
+// Count key pairs that a set of positions fails to distinguish. Keys whose
+// lengths differ modulo the table size are separated by the length term in the
+// hash, so they need no position coverage.
+template <std::size_t N>
+consteval std::size_t count_undistinguished_pairs(
+    const std::array<std::string_view, N>& keys,
+    const std::size_t* positions,
+    std::size_t num_positions,
+    std::size_t modulus) {
+    std::size_t count = 0;
+    for (std::size_t i = 0; i < N; ++i) {
+        for (std::size_t j = i + 1; j < N; ++j) {
+            if (keys[i].size() % modulus != keys[j].size() % modulus) { continue; }
+            bool distinguished = false;
+            for (std::size_t p = 0; p < num_positions; ++p) {
+                if (char_at(keys[i], positions[p]) != char_at(keys[j], positions[p])) {
+                    distinguished = true;
+                    break;
+                }
+            }
+            if (!distinguished) { ++count; }
+        }
+    }
+    return count;
+}
+
+template <std::size_t N>
+consteval bool positions_distinguish(
+    const std::array<std::string_view, N>& keys,
+    const std::size_t* positions,
+    std::size_t num_positions,
+    std::size_t modulus) {
+    return count_undistinguished_pairs<N>(keys, positions, num_positions, modulus) == 0;
+}
+
+// Number of distinct (length % modulus, char_at(key, pos)) pairs at a position.
+template <std::size_t N>
+consteval std::size_t discriminating_power(
+    const std::array<std::string_view, N>& keys,
+    std::size_t pos,
+    std::size_t modulus) {
+    struct pair { std::size_t len_mod; std::size_t ch; };
+    std::array<pair, N> pairs{};
+    for (std::size_t i = 0; i < N; ++i) {
+        pairs[i] = {keys[i].size() % modulus, char_at(keys[i], pos)};
+    }
+    std::size_t count = 0;
+    for (std::size_t i = 0; i < N; ++i) {
+        bool dup = false;
+        for (std::size_t j = 0; j < i; ++j) {
+            if (pairs[i].len_mod == pairs[j].len_mod && pairs[i].ch == pairs[j].ch) {
+                dup = true;
+                break;
+            }
+        }
+        if (!dup) { ++count; }
+    }
+    return count;
+}
+
+template <std::size_t N>
+consteval std::size_t max_key_length(const std::array<std::string_view, N>& keys) {
+    std::size_t m = 0;
+    for (std::size_t i = 0; i < N; ++i) {
+        if (keys[i].size() > m) { m = keys[i].size(); }
+    }
+    return m;
+}
+
+// Bounded backtracking DFS for a minimal set of distinguishing positions.
+template <std::size_t N>
+consteval bool backtracking_search(
+    const std::array<std::string_view, N>& keys,
+    const std::size_t* candidates,
+    std::size_t num_candidates,
+    std::size_t* positions,
+    std::size_t& num_positions_out,
+    std::size_t& budget,
+    std::size_t modulus) {
+    constexpr std::size_t MAX_DEPTH = 8;
+    std::size_t breadth = num_candidates < 20 ? num_candidates : 20;
+
+    struct frame { std::size_t depth; std::size_t next_ci; std::size_t parent_count; };
+    std::array<frame, MAX_DEPTH + 1> stack{};
+    std::size_t sp = 0;
+
+    std::size_t initial_count = count_undistinguished_pairs<N>(keys, positions, 0, modulus);
+    if (budget > 0) { --budget; }
+    if (initial_count == 0) { num_positions_out = 0; return true; }
+
+    stack[0] = {0, 0, initial_count};
+
+    while (budget > 0) {
+        if (sp > MAX_DEPTH) {
+            if (sp == 0) { break; }
+            --sp;
+            ++stack[sp].next_ci;
+            continue;
+        }
+        auto& f = stack[sp];
+        if (f.next_ci >= breadth) {
+            if (sp == 0) { break; }
+            --sp;
+            ++stack[sp].next_ci;
+            continue;
+        }
+        positions[sp] = candidates[f.next_ci];
+        --budget;
+        std::size_t new_count = count_undistinguished_pairs<N>(keys, positions, sp + 1, modulus);
+        if (new_count == 0) { num_positions_out = sp + 1; return true; }
+        if (new_count < f.parent_count && sp + 1 < MAX_DEPTH) {
+            stack[sp + 1] = {sp + 1, f.next_ci + 1, new_count};
+            ++sp;
+        } else {
+            ++f.next_ci;
+        }
+    }
+    return false;
+}
+
+// Phase 1: select character positions that distinguish all colliding pairs.
+template <std::size_t N>
+consteval std::size_t select_positions(
+    const std::array<std::string_view, N>& keys,
+    std::array<std::size_t, MAX_POSITIONS>& positions,
+    std::size_t modulus) {
+    if (positions_distinguish<N>(keys, positions.data(), 0, modulus)) { return 0; }
+
+    std::size_t max_len = max_key_length(keys);
+    constexpr std::size_t MAX_CANDIDATES = 256;
+    std::array<std::size_t, MAX_CANDIDATES> candidates{};
+    std::array<std::size_t, MAX_CANDIDATES> powers{};
+    std::size_t num_candidates = 0;
+    for (std::size_t p = 0; p < max_len && num_candidates < MAX_CANDIDATES - 1; ++p) {
+        candidates[num_candidates] = p;
+        powers[num_candidates] = discriminating_power(keys, p, modulus);
+        ++num_candidates;
+    }
+    if (num_candidates < MAX_CANDIDATES) {
+        candidates[num_candidates] = LAST_CHAR;
+        powers[num_candidates] = discriminating_power(keys, LAST_CHAR, modulus);
+        ++num_candidates;
+    }
+    for (std::size_t i = 0; i < num_candidates; ++i) {
+        for (std::size_t j = i + 1; j < num_candidates; ++j) {
+            if (powers[j] > powers[i]) {
+                auto tc = candidates[i]; candidates[i] = candidates[j]; candidates[j] = tc;
+                auto tp = powers[i]; powers[i] = powers[j]; powers[j] = tp;
+            }
+        }
+    }
+
+    positions[0] = candidates[0];
+    if (positions_distinguish<N>(keys, positions.data(), 1, modulus)) { return 1; }
+
+    positions[0] = 0;
+    positions[1] = LAST_CHAR;
+    if (positions_distinguish<N>(keys, positions.data(), 2, modulus)) { return 2; }
+
+    {
+        std::size_t budget = 5000;
+        std::size_t num_found = 0;
+        if (backtracking_search<N>(keys, candidates.data(), num_candidates,
+                                   positions.data(), num_found, budget, modulus)) {
+            return num_found;
+        }
+    }
+
+    std::size_t num_pos = 0;
+    for (std::size_t ci = 0; ci < num_candidates && num_pos < MAX_POSITIONS; ++ci) {
+        bool already = false;
+        for (std::size_t p = 0; p < num_pos; ++p) {
+            if (positions[p] == candidates[ci]) { already = true; break; }
+        }
+        if (already) { continue; }
+        positions[num_pos] = candidates[ci];
+        ++num_pos;
+        if (positions_distinguish<N>(keys, positions.data(), num_pos, modulus)) { return num_pos; }
+    }
+
+    compile_time_error("Failed to find distinguishing positions for perfect hash");
+    return 0;
+}
+
+// Result of PHF computation. A max-sized slot_to_key array lets the same struct
+// type carry any chosen table size.
+template <std::size_t N>
+struct phf_result {
+    // Allow up to 8x the minimum table size. Sparser tables solve faster.
+    static constexpr std::size_t MAX_TABLE_SIZE = next_power_of_2(N) * 8;
+    std::size_t table_size{};
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso_values{};
+    std::size_t num_positions{};
+    std::array<std::size_t, MAX_POSITIONS> positions{};
+    std::array<std::size_t, MAX_TABLE_SIZE> slot_to_key{};
+};
+
+// Partition-based asso_values search (gperf-style). Determines asso_values one
+// (position, character) symbol at a time; never revisits a value. Equivalence
+// classes (keys sharing the same undetermined symbols) keep the search cheap.
+template <std::size_t N, std::size_t M>
+consteval bool try_generate_gperf(
+    const std::array<std::string_view, N>& keys,
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS>& asso_values,
+    std::size_t& num_positions,
+    std::array<std::size_t, MAX_POSITIONS>& positions,
+    std::array<std::size_t, M>& slot_to_key) {
+    num_positions = select_positions<N>(keys, positions, M);
+
+    for (std::size_t p = 0; p < MAX_POSITIONS; ++p) {
+        for (std::size_t c = 0; c < 256; ++c) { asso_values[p][c] = 0; }
+    }
+
+    if (num_positions == 0) {
+        for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+        for (std::size_t i = 0; i < N; ++i) {
+            std::size_t slot = keys[i].size() % M;
+            if (slot_to_key[slot] != N) { return false; }
+            slot_to_key[slot] = i;
+        }
+        return true;
+    }
+
+    std::array<std::array<std::size_t, MAX_POSITIONS>, N> kchars{};
+    for (std::size_t k = 0; k < N; ++k) {
+        for (std::size_t p = 0; p < num_positions; ++p) {
+            kchars[k][p] = char_at(keys[k], positions[p]);
+        }
+    }
+
+    struct sym_t { std::size_t pos; std::size_t ch; std::size_t freq; };
+    constexpr std::size_t MAX_SYMS = MAX_POSITIONS * 256;
+    std::array<sym_t, MAX_SYMS> syms{};
+    std::size_t nsyms = 0;
+    for (std::size_t p = 0; p < num_positions; ++p) {
+        std::array<std::size_t, 256> freq{};
+        for (std::size_t k = 0; k < N; ++k) {
+            std::size_t c = kchars[k][p];
+            if (c < 256) { freq[c]++; }
+        }
+        for (std::size_t c = 0; c < 256; ++c) {
+            if (freq[c] > 0) { syms[nsyms++] = {p, c, freq[c]}; }
+        }
+    }
+    for (std::size_t i = 0; i < nsyms; ++i) {
+        for (std::size_t j = i + 1; j < nsyms; ++j) {
+            if (syms[j].freq > syms[i].freq) {
+                auto tmp = syms[i]; syms[i] = syms[j]; syms[j] = tmp;
+            }
+        }
+    }
+
+    std::array<std::size_t, N> phash{};
+    for (std::size_t k = 0; k < N; ++k) { phash[k] = keys[k].size(); }
+
+    std::array<std::array<uint64_t, 256>, MAX_POSITIONS> salt{};
+    {
+        uint64_t s = 0x9e3779b97f4a7c15ULL;
+        for (std::size_t p = 0; p < num_positions; ++p) {
+            for (std::size_t c = 0; c < 256; ++c) {
+                s = s * 6364136223846793005ULL + 1442695040888963407ULL;
+                salt[p][c] = s;
+            }
+        }
+    }
+    std::array<uint64_t, N> sig{};
+    for (std::size_t k = 0; k < N; ++k) {
+        uint64_t s = 0;
+        for (std::size_t p = 0; p < num_positions; ++p) {
+            std::size_t c = kchars[k][p];
+            if (c < 256) { s ^= salt[p][c]; }
+        }
+        sig[k] = s;
+    }
+    std::array<std::size_t, N> order{};
+    for (std::size_t k = 0; k < N; ++k) { order[k] = k; }
+
+    std::array<std::size_t, M> slot_gen{};
+    std::size_t gen = 0;
+
+    std::size_t search_limit = next_power_of_2(M);
+    if (search_limit < 32) { search_limit = 32; }
+
+    for (std::size_t si = 0; si < nsyms; ++si) {
+        std::size_t sp = syms[si].pos;
+        std::size_t sc = syms[si].ch;
+
+        uint64_t sp_salt = salt[sp][sc];
+        for (std::size_t k = 0; k < N; ++k) {
+            if (kchars[k][sp] == sc) { sig[k] ^= sp_salt; }
+        }
+
+        for (std::size_t i = 1; i < N; ++i) {
+            std::size_t x = order[i];
+            uint64_t xs = sig[x];
+            std::size_t j = i;
+            while (j > 0 && sig[order[j - 1]] > xs) {
+                order[j] = order[j - 1];
+                --j;
+            }
+            order[j] = x;
+        }
+
+        bool found = false;
+        for (std::size_t v = 0; v < search_limit && !found; ++v) {
+            bool collision = false;
+            std::size_t ci = 0;
+            while (ci < N && !collision) {
+                uint64_t class_sig = sig[order[ci]];
+                std::size_t cj = ci;
+                while (cj < N && sig[order[cj]] == class_sig) { ++cj; }
+                if (cj - ci > 1) {
+                    ++gen;
+                    for (std::size_t x = ci; x < cj; ++x) {
+                        std::size_t k = order[x];
+                        std::size_t h = phash[k];
+                        if (kchars[k][sp] == sc) { h += v; }
+                        h %= M;
+                        if (slot_gen[h] == gen) { collision = true; break; }
+                        slot_gen[h] = gen;
+                    }
+                }
+                ci = cj;
+            }
+            if (!collision) {
+                asso_values[sp][sc] = v;
+                for (std::size_t k = 0; k < N; ++k) {
+                    if (kchars[k][sp] == sc) { phash[k] += v; }
+                }
+                found = true;
+            }
+        }
+        if (!found) { return false; }
+    }
+
+    for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+    for (std::size_t i = 0; i < N; ++i) {
+        std::size_t slot = phash[i] % M;
+        if (slot_to_key[slot] != N) { return false; }
+        slot_to_key[slot] = i;
+    }
+    std::size_t filled = 0;
+    for (std::size_t i = 0; i < M; ++i) {
+        if (slot_to_key[i] != N) { ++filled; }
+    }
+    return filled == N;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_compute_phf(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+    static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso{};
+    std::size_t npos{};
+    std::array<std::size_t, MAX_POSITIONS> pos{};
+    std::array<std::size_t, M> s2k{};
+    if (try_generate_gperf<N, M>(keys, asso, npos, pos, s2k)) {
+        result.table_size = M;
+        result.asso_values = asso;
+        result.num_positions = npos;
+        result.positions = pos;
+        for (std::size_t i = 0; i < M; ++i) { result.slot_to_key[i] = s2k[i]; }
+        for (std::size_t i = M; i < phf_result<N>::MAX_TABLE_SIZE; ++i) { result.slot_to_key[i] = N; }
+        return true;
+    }
+    return false;
+}
+
+template <std::size_t N, std::size_t M, std::size_t MaxM>
+consteval bool try_gperf_po2(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+    if (try_compute_phf<N, M>(keys, result)) { return true; }
+    constexpr std::size_t NextM = M * 2;
+    if constexpr (NextM <= MaxM) { return try_gperf_po2<N, NextM, MaxM>(keys, result); }
+    return false;
+}
+
+// --- Hash-and-Displace fallback --------------------------------------------
+
+constexpr std::size_t hd_bucket_hash(std::string_view key) noexcept {
+    std::size_t c0 = key.empty() ? 0 : static_cast<unsigned char>(key[0]);
+    std::size_t c1 = key.empty() ? 0 : static_cast<unsigned char>(key[key.size() - 1]);
+    return (c0 + c1 * 3 + key.size() * 17) & 0xFF;
+}
+constexpr std::size_t hd_safe_char(const char* p, std::size_t len, std::size_t idx) noexcept {
+    std::size_t has = static_cast<std::size_t>(idx < len);
+    std::size_t si = idx & (std::size_t{0} - has);
+    return static_cast<unsigned char>(p[si]) & (std::size_t{0} - has);
+}
+constexpr std::size_t hd_key_hash_2(std::string_view key) noexcept {
+    std::size_t kc = key.size();
+    kc = kc * 31 + static_cast<unsigned char>(key[0]);
+    kc = kc * 31 + hd_safe_char(key.data(), key.size(), 1);
+    return kc;
+}
+constexpr std::size_t hd_key_hash_4(std::string_view key) noexcept {
+    std::size_t kc = key.size();
+    kc = kc * 31 + static_cast<unsigned char>(key[0]);
+    kc = kc * 31 + hd_safe_char(key.data(), key.size(), 1);
+    kc = kc * 31 + hd_safe_char(key.data(), key.size(), 2);
+    kc = kc * 31 + hd_safe_char(key.data(), key.size(), 3);
+    return kc;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_hash_and_displace(
+    const std::array<std::string_view, N>& keys,
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS>& asso_values,
+    std::size_t& num_positions,
+    std::array<std::size_t, MAX_POSITIONS>& positions,
+    std::array<std::size_t, M>& slot_to_key) {
+    num_positions = select_positions<N>(keys, positions, M);
+
+    if (num_positions == 0) {
+        for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+        for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+        for (std::size_t i = 0; i < N; ++i) {
+            std::size_t slot = keys[i].size() % M;
+            if (slot_to_key[slot] != N) { return false; }
+            slot_to_key[slot] = i;
+        }
+        return true;
+    }
+
+    for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+    num_positions = HD_MODE; // sentinel for H&D mode
+    positions[0] = 0;
+    positions[1] = LAST_CHAR;
+
+    std::array<std::size_t, N> key_bucket{};
+    for (std::size_t i = 0; i < N; ++i) { key_bucket[i] = hd_bucket_hash(keys[i]); }
+
+    struct bucket_info { std::size_t ch; std::size_t count; };
+    std::array<bucket_info, N> buckets{};
+    std::size_t num_buckets = 0;
+    for (std::size_t i = 0; i < N; ++i) {
+        std::size_t bk = key_bucket[i];
+        bool found = false;
+        for (std::size_t b = 0; b < num_buckets; ++b) {
+            if (buckets[b].ch == bk) { ++buckets[b].count; found = true; break; }
+        }
+        if (!found) { buckets[num_buckets++] = {bk, 1}; }
+    }
+    for (std::size_t i = 0; i < num_buckets; ++i) {
+        for (std::size_t j = i + 1; j < num_buckets; ++j) {
+            if (buckets[j].count > buckets[i].count) {
+                auto tmp = buckets[i]; buckets[i] = buckets[j]; buckets[j] = tmp;
+            }
+        }
+    }
+
+    auto try_placement = [&](auto key_hash_fn) -> bool {
+        for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+        for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+        for (std::size_t b = 0; b < num_buckets; ++b) {
+            std::size_t ch = buckets[b].ch;
+            std::array<std::size_t, N> bucket_keys{};
+            std::size_t bk_count = 0;
+            for (std::size_t i = 0; i < N; ++i) {
+                if (key_bucket[i] == ch) { bucket_keys[bk_count++] = i; }
+            }
+            bool placed = false;
+            std::size_t max_d = M < 255 ? M : 255;
+            for (std::size_t d = 0; d < max_d; ++d) {
+                bool ok = true;
+                std::array<std::size_t, N> bucket_slots{};
+                for (std::size_t k = 0; k < bk_count; ++k) {
+                    std::size_t slot = (d + key_hash_fn(keys[bucket_keys[k]])) % M;
+                    if (slot_to_key[slot] != N) { ok = false; break; }
+                    for (std::size_t k2 = 0; k2 < k; ++k2) {
+                        if (bucket_slots[k2] == slot) { ok = false; break; }
+                    }
+                    if (!ok) { break; }
+                    bucket_slots[k] = slot;
+                }
+                if (ok) {
+                    asso_values[0][ch] = d;
+                    for (std::size_t k = 0; k < bk_count; ++k) {
+                        slot_to_key[bucket_slots[k]] = bucket_keys[k];
+                    }
+                    placed = true;
+                    break;
+                }
+            }
+            if (!placed) { return false; }
+        }
+        std::size_t filled = 0;
+        for (std::size_t i = 0; i < M; ++i) {
+            if (slot_to_key[i] != N) { ++filled; }
+        }
+        return filled == N;
+    };
+
+    if (try_placement([](std::string_view k) { return hd_key_hash_2(k); })) {
+        positions[2] = HD_HASH_2BYTE_FLAG;
+        return true;
+    }
+    if (try_placement([](std::string_view k) { return hd_key_hash_4(k); })) {
+        positions[2] = HD_HASH_4BYTE_FLAG;
+        return true;
+    }
+    return false;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_compute_phf_hd(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+    static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso{};
+    std::size_t npos{};
+    std::array<std::size_t, MAX_POSITIONS> pos{};
+    std::array<std::size_t, M> s2k{};
+    if (try_hash_and_displace<N, M>(keys, asso, npos, pos, s2k)) {
+        result.table_size = M;
+        result.asso_values = asso;
+        result.num_positions = npos;
+        result.positions = pos;
+        for (std::size_t i = 0; i < M; ++i) { result.slot_to_key[i] = s2k[i]; }
+        for (std::size_t i = M; i < phf_result<N>::MAX_TABLE_SIZE; ++i) { result.slot_to_key[i] = N; }
+        return true;
+    }
+    return false;
+}
+
+template <std::size_t N, std::size_t M>
+consteval phf_result<N> compute_phf_hd_po2(const std::array<std::string_view, N>& keys) {
+    static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+    phf_result<N> result{};
+    if (try_compute_phf_hd<N, M>(keys, result)) { return result; }
+    constexpr std::size_t NextM = M * 2;
+    if constexpr (NextM <= phf_result<N>::MAX_TABLE_SIZE) {
+        return compute_phf_hd_po2<N, NextM>(keys);
+    } else {
+        compile_time_error("Hash-and-Displace: failed to find valid table size");
+        return result;
+    }
+}
+
+// Compute a perfect hash for `keys`: try gperf at power-of-two sizes (capped so
+// the runtime tables stay within uint8 indices), then fall back to H&D.
+template <std::size_t N>
+consteval phf_result<N> compute_phf(const std::array<std::string_view, N>& keys) {
+    constexpr std::size_t StartM = next_power_of_2(N);
+    constexpr std::size_t GPERF_MAX_TABLE =
+        phf_result<N>::MAX_TABLE_SIZE < 256 ? phf_result<N>::MAX_TABLE_SIZE : 256;
+    if constexpr (StartM <= GPERF_MAX_TABLE) {
+        phf_result<N> result{};
+        if (try_gperf_po2<N, StartM, GPERF_MAX_TABLE>(keys, result)) { return result; }
+    }
+    return compute_phf_hd_po2<N, StartM>(keys);
+}
+
+// ============================================================================
+// Runtime tables (flat, uint8) derived from a phf_result.
+// ============================================================================
+
+template <std::size_t N, std::size_t TableSize, std::size_t MaxKeyLen>
+struct phf_data {
+    std::array<std::array<std::uint8_t, 256>, MAX_POSITIONS> asso_values{};
+    std::array<std::uint8_t, MAX_POSITIONS>                  positions{};
+    std::uint8_t                                             num_positions{};
+    std::uint8_t                                             hd_hash_variant{}; // 2 or 4 (H&D only)
+    std::array<std::uint8_t, TableSize>                      slot_to_key{};
+    // slot_key_bytes[s] holds the key stored at slot s, zero-padded to a 16-byte
+    // multiple so the SIMD comparison can read a whole register.
+    std::array<std::array<char, ((MaxKeyLen + 15) / 16) * 16>, TableSize> slot_key_bytes{};
+    std::array<std::uint8_t, TableSize>                      slot_key_len{};
+};
+
+template <std::size_t N>
+constexpr std::size_t compute_max_key_len(const std::array<std::string_view, N>& keys) noexcept {
+    std::size_t m = 0;
+    for (std::size_t i = 0; i < N; ++i) { if (keys[i].size() > m) { m = keys[i].size(); } }
+    return m;
+}
+
+// Validate keys and build the runtime tables from the computed perfect hash.
+template <std::size_t N, std::size_t TableSize, std::size_t MaxKeyLen>
+consteval phf_data<N, TableSize, MaxKeyLen>
+build_phf_data(const std::array<std::string_view, N>& keys, const phf_result<N>& result) {
+    for (std::size_t i = 0; i < N; ++i) {
+        if (keys[i].empty())            { compile_time_error("empty keys are not allowed in key_selector"); }
+        if (keys[i].size() > MaxKeyLen) { compile_time_error("key length exceeds MaxKeyLen"); }
+        for (char c : keys[i]) {
+            if (c == '\\') { compile_time_error("backslash not allowed in key_selector keys"); }
+            if (c == '"')  { compile_time_error("quote not allowed in key_selector keys"); }
+            if (c == '\0') { compile_time_error("null byte not allowed in key_selector keys"); }
+        }
+        for (std::size_t j = i + 1; j < N; ++j) {
+            if (keys[i] == keys[j]) { compile_time_error("duplicate keys in key_selector"); }
+        }
+    }
+
+    phf_data<N, TableSize, MaxKeyLen> out{};
+
+    if (result.num_positions == HD_MODE) {
+        // H&D mode: single displacement table in asso_values[0].
+        for (std::size_t c = 0; c < 256; ++c) {
+            out.asso_values[0][c] = static_cast<std::uint8_t>(result.asso_values[0][c]);
+        }
+        out.num_positions   = static_cast<std::uint8_t>(HD_MODE);
+        out.hd_hash_variant = static_cast<std::uint8_t>(result.positions[2]);
+    } else {
+        for (std::size_t pi = 0; pi < result.num_positions; ++pi) {
+            for (std::size_t c = 0; c < 256; ++c) {
+                out.asso_values[pi][c] = static_cast<std::uint8_t>(result.asso_values[pi][c] % TableSize);
+            }
+        }
+        out.num_positions = static_cast<std::uint8_t>(result.num_positions);
+        for (std::size_t i = 0; i < result.num_positions; ++i) {
+            out.positions[i] = (result.positions[i] == LAST_CHAR)
+                ? POS_LAST_CHAR
+                : static_cast<std::uint8_t>(result.positions[i]);
+        }
+    }
+
+    for (std::size_t s = 0; s < TableSize; ++s) {
+        out.slot_to_key[s] = static_cast<std::uint8_t>(result.slot_to_key[s]);
+    }
+
+    for (std::size_t s = 0; s < TableSize; ++s) {
+        std::size_t ki = result.slot_to_key[s];
+        if (ki < N) {
+            auto k = keys[ki];
+            out.slot_key_len[s] = static_cast<std::uint8_t>(k.size());
+            for (std::size_t c = 0; c < k.size(); ++c) { out.slot_key_bytes[s][c] = k[c]; }
+        } else {
+            out.slot_key_len[s] = 0; // empty slot: no length can match
+        }
+    }
+    return out;
+}
+
+// --- SIMD runtime primitives ------------------------------------------------
+
+// The runtime matchers read whole 16-byte blocks up to offset 63 (four blocks)
+// for keys as long as the 63-character maximum. That read must stay within the
+// buffer's trailing padding, so the padding has to exceed the largest offset we
+// touch.
+static_assert(SIMDJSON_PADDING > 63,
+              "key_selector requires SIMDJSON_PADDING > 63 for its SIMD key reads");
+
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+// True when every byte of `diff` is zero. A 32-bit-lane horizontal max is
+// enough (zero iff every 32-bit word is zero) and is cheaper than a byte-wide
+// reduction. Do not replace this with a floating-point compare against 0.0:
+// flush-to-zero would treat a denormal as zero.
+simdjson_really_inline bool neon_all_bytes_zero(uint8x16_t diff) noexcept {
+    return vmaxvq_u32(vreinterpretq_u32_u8(diff)) == 0;
+}
+#endif
+
+// Scan for the terminating '"' starting at p. Returns its byte offset (= key
+// length). Caller guarantees SIMDJSON_PADDING (== 64) bytes past the JSON buffer,
+// so reading whole 16-byte blocks up to offset 63 is always safe.
+template <std::size_t MaxKeyLen>
+simdjson_really_inline std::size_t scan_key_length(const char* p) noexcept {
+    // The closing quote of the longest key sits at most at offset MaxKeyLen, so we
+    // scan as many 16-byte blocks as it takes to cover offsets 0..MaxKeyLen. The
+    // cap (63) keeps every covered offset within the 64-byte padding guarantee, so
+    // the SIMD and scalar builds agree.
+    static_assert(MaxKeyLen <= 63, "MaxKeyLen must be <= 63 for current SIMD implementations");
+    // [[maybe_unused]]: the scalar fallback (no NEON/SSE2/LSX) does not use it.
+    [[maybe_unused]] constexpr std::size_t num_blocks = MaxKeyLen / 16 + 1;
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+    for (std::size_t b = 0; b < num_blocks; ++b) {
+        uint8x16_t v = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + b * 16);
+        uint8x16_t cmp = vceqq_u8(v, vdupq_n_u8('"'));
+        uint64_t m = vget_lane_u64(
+            vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(cmp), 4)), 0);
+        if (simdjson_likely(m != 0)) { return b * 16 + (std::size_t(__builtin_ctzll(m)) >> 2); }
+    }
+    return MaxKeyLen + 1;
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+    for (std::size_t b = 0; b < num_blocks; ++b) {
+        __m128i v   = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + b * 16));
+        __m128i cmp = _mm_cmpeq_epi8(v, _mm_set1_epi8('"'));
+        unsigned m  = static_cast<unsigned>(_mm_movemask_epi8(cmp));
+        if (simdjson_likely(m != 0)) { return b * 16 + std::size_t(__builtin_ctz(m)); }
+    }
+    return MaxKeyLen + 1;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+    for (std::size_t b = 0; b < num_blocks; ++b) {
+        __m128i v   = __lsx_vld(reinterpret_cast<const void*>(p + b * 16), 0);
+        __m128i cmp = __lsx_vseq_b(v, __lsx_vreplgr2vr_b('"'));
+        // vmskltz_b gathers the per-byte sign bits (set where the byte equals '"')
+        // into the low 16 bits of lane 0, the LSX equivalent of movemask.
+        unsigned m  = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(cmp), 0)) & 0xFFFFu;
+        if (simdjson_likely(m != 0)) { return b * 16 + std::size_t(__builtin_ctz(m)); }
+    }
+    return MaxKeyLen + 1;
+#else
+    for (std::size_t i = 0; i <= MaxKeyLen; ++i)
+        if (p[i] == '"') return i;
+    return MaxKeyLen + 1;
+#endif
+}
+
+// Byte-equal of p[0..len) against stored[0..len). stored is zero-padded past
+// `len`. Input is read over 16, 32, 48, or 64 bytes (padded JSON buffer
+// guaranteed; SIMDJSON_PADDING == 64).
+template <std::size_t MaxKeyLen>
+simdjson_really_inline bool compare_key_bytes(
+    const char* p, const char* stored, std::size_t len) noexcept {
+    // [[maybe_unused]]: only the NEON/SSE2/LSX branches read these; the scalar
+    // fallback build (no SIMD) leaves them unused, which is an error under -Werror.
+    [[maybe_unused]] alignas(16) static constexpr uint8_t idx16[16] =
+        {0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15};
+    if constexpr (MaxKeyLen <= 16) {
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+        uint8x16_t vp = vld1q_u8(reinterpret_cast<const uint8_t*>(p));
+        uint8x16_t vs = vld1q_u8(reinterpret_cast<const uint8_t*>(stored));
+        uint8x16_t mask = vcltq_u8(vld1q_u8(idx16), vdupq_n_u8(static_cast<uint8_t>(len)));
+        uint8x16_t diff = veorq_u8(vandq_u8(vp, mask), vs);
+        return neon_all_bytes_zero(diff);
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+        __m128i vp = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p));
+        __m128i vs = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored));
+        __m128i idx = _mm_load_si128(reinterpret_cast<const __m128i*>(idx16));
+        __m128i mask = _mm_cmplt_epi8(idx, _mm_set1_epi8(static_cast<char>(len)));
+        __m128i eq = _mm_cmpeq_epi8(_mm_and_si128(vp, mask), vs);
+        return _mm_movemask_epi8(eq) == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+        __m128i vp = __lsx_vld(reinterpret_cast<const void*>(p), 0);
+        __m128i vs = __lsx_vld(reinterpret_cast<const void*>(stored), 0);
+        __m128i idx = __lsx_vld(reinterpret_cast<const void*>(idx16), 0);
+        __m128i mask = __lsx_vslt_b(idx, __lsx_vreplgr2vr_b(static_cast<int>(len)));
+        __m128i eq = __lsx_vseq_b(__lsx_vand_v(vp, mask), vs);
+        return (static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq), 0)) & 0xFFFFu) == 0xFFFFu;
+#else
+        for (std::size_t i = 0; i < len; ++i)
+            if (p[i] != stored[i]) return false;
+        return true;
+#endif
+    } else if constexpr (MaxKeyLen <= 32) {
+        [[maybe_unused]] alignas(16) static constexpr uint8_t idx32_hi[16] =
+            {16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31};
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+        uint8x16_t vp_lo = vld1q_u8(reinterpret_cast<const uint8_t*>(p));
+        uint8x16_t vp_hi = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + 16);
+        uint8x16_t vs_lo = vld1q_u8(reinterpret_cast<const uint8_t*>(stored));
+        uint8x16_t vs_hi = vld1q_u8(reinterpret_cast<const uint8_t*>(stored) + 16);
+        uint8x16_t lenv = vdupq_n_u8(static_cast<uint8_t>(len));
+        uint8x16_t m_lo = vcltq_u8(vld1q_u8(idx16),    lenv);
+        uint8x16_t m_hi = vcltq_u8(vld1q_u8(idx32_hi), lenv);
+        uint8x16_t d_lo = veorq_u8(vandq_u8(vp_lo, m_lo), vs_lo);
+        uint8x16_t d_hi = veorq_u8(vandq_u8(vp_hi, m_hi), vs_hi);
+        return neon_all_bytes_zero(vorrq_u8(d_lo, d_hi));
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+        __m128i vp_lo = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p));
+        __m128i vp_hi = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + 16));
+        __m128i vs_lo = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored));
+        __m128i vs_hi = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored + 16));
+        __m128i lenv = _mm_set1_epi8(static_cast<char>(len));
+        __m128i m_lo = _mm_cmplt_epi8(_mm_load_si128(reinterpret_cast<const __m128i*>(idx16)),    lenv);
+        __m128i m_hi = _mm_cmplt_epi8(_mm_load_si128(reinterpret_cast<const __m128i*>(idx32_hi)), lenv);
+        __m128i eq_lo = _mm_cmpeq_epi8(_mm_and_si128(vp_lo, m_lo), vs_lo);
+        __m128i eq_hi = _mm_cmpeq_epi8(_mm_and_si128(vp_hi, m_hi), vs_hi);
+        return (_mm_movemask_epi8(eq_lo) & _mm_movemask_epi8(eq_hi)) == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+        __m128i lenv = __lsx_vreplgr2vr_b(static_cast<int>(len));
+        __m128i vp_lo = __lsx_vld(reinterpret_cast<const void*>(p), 0);
+        __m128i vp_hi = __lsx_vld(reinterpret_cast<const void*>(p + 16), 0);
+        __m128i vs_lo = __lsx_vld(reinterpret_cast<const void*>(stored), 0);
+        __m128i vs_hi = __lsx_vld(reinterpret_cast<const void*>(stored + 16), 0);
+        __m128i m_lo = __lsx_vslt_b(__lsx_vld(reinterpret_cast<const void*>(idx16), 0),    lenv);
+        __m128i m_hi = __lsx_vslt_b(__lsx_vld(reinterpret_cast<const void*>(idx32_hi), 0), lenv);
+        __m128i eq_lo = __lsx_vseq_b(__lsx_vand_v(vp_lo, m_lo), vs_lo);
+        __m128i eq_hi = __lsx_vseq_b(__lsx_vand_v(vp_hi, m_hi), vs_hi);
+        unsigned mlo = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq_lo), 0)) & 0xFFFFu;
+        unsigned mhi = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq_hi), 0)) & 0xFFFFu;
+        return (mlo & mhi) == 0xFFFFu;
+#else
+        for (std::size_t i = 0; i < len; ++i)
+            if (p[i] != stored[i]) return false;
+        return true;
+#endif
+    } else if constexpr (MaxKeyLen <= 64) {
+        // 3 or 4 16-byte blocks (33..48 -> 3, 49..64 -> 4). stored is zero-padded
+        // to exactly num_blocks*16 bytes (KEY_STRIDE), so neither load overruns it.
+        // [[maybe_unused]]: the scalar fallback (no NEON/SSE2/LSX) does not use it.
+        [[maybe_unused]] constexpr std::size_t num_blocks = (MaxKeyLen + 15) / 16;
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+        uint8x16_t base = vld1q_u8(idx16);
+        uint8x16_t lenv = vdupq_n_u8(static_cast<uint8_t>(len));
+        uint8x16_t acc  = vdupq_n_u8(0);
+        for (std::size_t b = 0; b < num_blocks; ++b) {
+            uint8x16_t vp   = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + b * 16);
+            uint8x16_t vs   = vld1q_u8(reinterpret_cast<const uint8_t*>(stored) + b * 16);
+            uint8x16_t idxv = vaddq_u8(base, vdupq_n_u8(static_cast<uint8_t>(b * 16)));
+            uint8x16_t mask = vcltq_u8(idxv, lenv);
+            acc = vorrq_u8(acc, veorq_u8(vandq_u8(vp, mask), vs));
+        }
+        return neon_all_bytes_zero(acc);
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+        __m128i base = _mm_load_si128(reinterpret_cast<const __m128i*>(idx16));
+        __m128i lenv = _mm_set1_epi8(static_cast<char>(len));
+        int eq = 0xFFFF;
+        for (std::size_t b = 0; b < num_blocks; ++b) {
+            __m128i vp   = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + b * 16));
+            __m128i vs   = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored + b * 16));
+            __m128i idxv = _mm_add_epi8(base, _mm_set1_epi8(static_cast<char>(b * 16)));
+            __m128i mask = _mm_cmplt_epi8(idxv, lenv);
+            eq &= _mm_movemask_epi8(_mm_cmpeq_epi8(_mm_and_si128(vp, mask), vs));
+        }
+        return eq == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+        __m128i base = __lsx_vld(reinterpret_cast<const void*>(idx16), 0);
+        __m128i lenv = __lsx_vreplgr2vr_b(static_cast<int>(len));
+        unsigned acc = 0xFFFFu;
+        for (std::size_t b = 0; b < num_blocks; ++b) {
+            __m128i vp   = __lsx_vld(reinterpret_cast<const void*>(p + b * 16), 0);
+            __m128i vs   = __lsx_vld(reinterpret_cast<const void*>(stored + b * 16), 0);
+            __m128i idxv = __lsx_vadd_b(base, __lsx_vreplgr2vr_b(static_cast<int>(b * 16)));
+            __m128i mask = __lsx_vslt_b(idxv, lenv);
+            __m128i eq   = __lsx_vseq_b(__lsx_vand_v(vp, mask), vs);
+            acc &= static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq), 0)) & 0xFFFFu;
+        }
+        return acc == 0xFFFFu;
+#else
+        for (std::size_t i = 0; i < len; ++i)
+            if (p[i] != stored[i]) return false;
+        return true;
+#endif
+    } else {
+        for (std::size_t i = 0; i < len; ++i)
+            if (p[i] != stored[i]) return false;
+        return true;
+    }
+}
+
+// --- Single 8-bit window fast path ------------------------------------------
+//
+// Many small key sets can be told apart by inspecting a *single* 8-bit window of
+// the key bytes -- and that window need not be byte-aligned. Because every JSON
+// key is terminated by a '"', the bytes at and before a key's length are well
+// defined for any key at least that long: byte i is the key character when i is
+// inside the key and the closing quote when i == len. So we read two bytes at a
+// fixed offset, extract 8 consecutive bits at a fixed intra-byte shift, and if
+// that value is distinct for every key, a 256-entry table maps it straight to a
+// candidate key. The match then needs no hash and -- in the length-free overload
+// -- no SIMD length scan: load two bytes, shift, mask, index the table, and
+// confirm the candidate with one comparison.
+//
+// Allowing an *unaligned* window (a shift of 1..7) mixes bits from two adjacent
+// bytes and discriminates key sets that no single aligned byte can. For example,
+// the partial_tweets keys {created_at,id,text,in_reply_to_status_id,user,
+// retweet_count,favorite_count} share a colliding byte at every aligned position
+// 0,1,2, yet the 8 bits starting at bit offset 2 are unique across all seven.
+// Simpler cases fall out as the shift==0 special case: {"id","screen_name"}
+// splits on byte 0, {"jo","joe"} on the quote at byte 2.
+//
+// The window is confined to the first (shortest key length + 1) bytes so the
+// two-byte read never crosses a key's closing quote into uncontrolled value
+// bytes; that final byte is the shortest key's quote.
+template <std::size_t N, std::size_t MaxKeyLen>
+struct window_data {
+    static constexpr std::size_t KEY_STRIDE = ((MaxKeyLen + 15) / 16) * 16;
+    bool                                            ok{false};
+    std::uint8_t                                    byte_offset{0}; // first byte of the 2-byte read
+    std::uint8_t                                    shift{0};       // intra-byte bit shift (0..7)
+    std::array<std::uint8_t, 256>                   window_to_key{}; // window byte -> key index, N if none
+    std::array<std::uint8_t, N>                     key_len{};
+    std::array<std::array<char, KEY_STRIDE>, N>     key_bytes{};     // zero-padded
+};
+
+// Byte seen at `idx` for key `i`: a key character when idx is inside the key, or
+// the closing '"' when idx == len (callers keep idx <= every key's length).
+template <std::size_t N>
+constexpr unsigned window_byte_at(const std::array<std::string_view, N>& keys,
+                                  std::size_t i, std::size_t idx) noexcept {
+    if (idx < keys[i].size()) { return static_cast<unsigned char>(keys[i][idx]); }
+    return static_cast<unsigned>('"');
+}
+
+// The 8-bit window value for key `i` at (byte_offset, shift): the little-endian
+// pair (byte[off], byte[off+1]) shifted right by `shift` and truncated. Matches
+// the runtime read exactly.
+template <std::size_t N>
+constexpr unsigned window_value(const std::array<std::string_view, N>& keys,
+                                std::size_t i, std::size_t byte_offset, std::size_t shift) noexcept {
+    unsigned lo = window_byte_at<N>(keys, i, byte_offset);
+    unsigned hi = window_byte_at<N>(keys, i, byte_offset + 1);
+    return ((lo | (hi << 8)) >> shift) & 0xFFu;
+}
+
+template <std::size_t N, std::size_t MaxKeyLen>
+consteval window_data<N, MaxKeyLen>
+compute_window(const std::array<std::string_view, N>& keys) {
+    window_data<N, MaxKeyLen> out{};
+
+    std::size_t min_len = keys[0].size();
+    for (std::size_t i = 1; i < N; ++i) {
+        if (keys[i].size() < min_len) { min_len = keys[i].size(); }
+    }
+
+    // Iterate windows nearest the front first (cheapest to read, smallest shift).
+    for (std::size_t off = 0; off <= min_len; ++off) {
+        for (std::size_t shift = 0; shift < 8; ++shift) {
+            // The read touches byte off, and byte off+1 when shift != 0. Both must
+            // stay within the safe region [0, min_len] (min_len is the shortest
+            // key's quote index). off <= min_len is guaranteed by the loop bound.
+            if (shift != 0 && off + 1 > min_len) { continue; }
+
+            bool distinct = true;
+            for (std::size_t i = 0; i < N && distinct; ++i) {
+                for (std::size_t j = i + 1; j < N; ++j) {
+                    if (window_value<N>(keys, i, off, shift) == window_value<N>(keys, j, off, shift)) {
+                        distinct = false;
+                        break;
+                    }
+                }
+            }
+            if (!distinct) { continue; }
+
+            out.ok          = true;
+            out.byte_offset = static_cast<std::uint8_t>(off);
+            out.shift       = static_cast<std::uint8_t>(shift);
+            for (std::size_t b = 0; b < 256; ++b) { out.window_to_key[b] = static_cast<std::uint8_t>(N); }
+            for (std::size_t i = 0; i < N; ++i) {
+                out.window_to_key[window_value<N>(keys, i, off, shift)] = static_cast<std::uint8_t>(i);
+                out.key_len[i] = static_cast<std::uint8_t>(keys[i].size());
+                for (std::size_t c = 0; c < keys[i].size(); ++c) { out.key_bytes[i][c] = keys[i][c]; }
+            }
+            return out;
+        }
+    }
+    return out; // ok == false: no single 8-bit window distinguishes the keys
+}
+
+// Read the 8-bit window at (byte_offset, shift) from a padded key pointer.
+simdjson_really_inline std::uint8_t read_window(const char* p, std::size_t byte_offset,
+                                                std::size_t shift) noexcept {
+    std::uint16_t w;
+    // Two controlled bytes (within the shortest key + its quote, hence within the
+    // padded buffer). memcpy is the portable little-endian unaligned load.
+    std::memcpy(&w, p + byte_offset, sizeof(w));
+#if defined(__BYTE_ORDER__) && (__BYTE_ORDER__ == __ORDER_BIG_ENDIAN__)
+    w = static_cast<std::uint16_t>((w >> 8) | (w << 8));
+#endif
+    return static_cast<std::uint8_t>((w >> shift) & 0xFFu);
+}
+
+// Verify a window candidate. The window table already mapped the key to the
+// single possible index `ki` (< N); here we confirm it. Folding over 0..N-1
+// turns the runtime `ki` into a compile-time index in the matching arm, so the
+// candidate's length and bytes are constants for compare_key_bytes -- the same
+// specialization the ordered find_field path gets from a CT-length
+// unsafe_is_equal, and what keeps this path competitive. The closing-quote check
+// (p[L] == '"') both confirms the key ends exactly at the candidate's length and
+// rejects a wrong-length key, so this works whether or not the caller knew len.
+template <std::size_t N, std::size_t MaxKeyLen, std::size_t... Is>
+simdjson_really_inline std::size_t
+match_window_candidate(const char* p, std::uint8_t ki,
+                       const window_data<N, MaxKeyLen>& w,
+                       std::index_sequence<Is...>) noexcept {
+  std::size_t result = N;
+  auto try_match = [&](auto Ic) {
+    constexpr std::size_t i = decltype(Ic)::value;
+    if (ki == i && p[w.key_len[i]] == '"' &&
+        key_selector_detail::compare_key_bytes<MaxKeyLen>(
+            p, w.key_bytes[i].data(), w.key_len[i])) {
+      result = i;
+    }
+  };
+  (try_match(std::integral_constant<std::size_t, Is>{}), ...);
+  return result;
+}
+
+// --- describe() string helpers (constexpr; no std::to_string, which is not) ---
+
+// Append the decimal form of v to s.
+SIMDJSON_CONSTEXPR_STRING void append_uint(std::string& s, std::size_t v) {
+    if (v == 0) { s.push_back('0'); return; }
+    char buf[20];
+    std::size_t n = 0;
+    while (v > 0) { buf[n++] = static_cast<char>('0' + (v % 10)); v /= 10; }
+    while (n > 0) { s.push_back(buf[--n]); }
+}
+
+// Append a byte as its decimal value, plus the printable character in quotes
+// when it is in the printable ASCII range (e.g. "110 ('n')").
+SIMDJSON_CONSTEXPR_STRING void append_byte(std::string& s, unsigned b) {
+    append_uint(s, b);
+    if (b >= 0x20 && b < 0x7f) {
+        s += " ('";
+        s.push_back(static_cast<char>(b));
+        s += "')";
+    }
+}
+
+} // namespace key_selector_detail
+
+/**
+ * Stateless, compile-time key selector.
+ *
+ * Usage:
+ *   using sel_t = key_selector<"id", "text", "user">;
+ *   std::size_t i = sel_t::match_raw(raw_key); // returns sel_t::size() on miss
+ *
+ * The perfect hash is built at compile time (gperf-style, with a
+ * Hash-and-Displace fallback) and only flat tables survive to runtime. All
+ * tables are static constexpr, so the lookup fully inlines.
+ *
+ * Limitations:
+ *   - Each key must be at most 63 characters long (and no longer than
+ *     SIMDJSON_PADDING). Longer keys trigger a compile-time error.
+ *   - The number of keys should be moderate. The hard limit is 255 keys;
+ *     compilation time grows with the number of keys, so prefer a few dozen at
+ *     most per selector.
+ *   - Keys must be distinct, non-empty, and free of backslash, double-quote and
+ *     null bytes (matching is done against the raw, unescaped JSON key bytes).
+ */
+template <constevalutil::fixed_string... Keys>
+struct key_selector {
+    static constexpr std::size_t N = sizeof...(Keys);
+    static_assert(N > 0,   "key_selector requires at least one key");
+    static_assert(N <= 255,"key_selector supports at most 255 keys");
+
+    static constexpr std::array<std::string_view, N> keys{ Keys.view()... };
+    static constexpr std::size_t max_key_len = key_selector_detail::compute_max_key_len<N>(keys);
+    static_assert(max_key_len <= SIMDJSON_PADDING,
+                  "key longer than SIMDJSON_PADDING is not supported");
+    // The SIMD key-length scan covers offsets 0..63 (four 16-byte blocks), which
+    // stays within the 64-byte padding guarantee. A 64-character key's closing
+    // quote would land at offset 64 and be missed on NEON/SSE2/LSX while still
+    // matching in scalar builds, so cap at 63 to keep implementations in agreement.
+    static_assert(max_key_len <= 63,
+                  "key_selector keys must be at most 63 characters long");
+
+    static constexpr auto result = key_selector_detail::compute_phf<N>(keys);
+    static constexpr std::size_t table_size = result.table_size;
+
+    static constexpr auto phf =
+        key_selector_detail::build_phf_data<N, table_size, max_key_len>(keys, result);
+
+    // Single 8-bit-window discriminator (when one exists). Detected at compile
+    // time and selected with `if constexpr` below, so the hash path is compiled
+    // out for key sets that qualify, and this is compiled out for those that do
+    // not.
+    static constexpr auto window =
+        key_selector_detail::compute_window<N, max_key_len>(keys);
+
+    static constexpr std::size_t size() noexcept { return N; }
+
+    /**
+     * Look up a JSON key whose length is already known. p must point at the first
+     * key byte (just after the opening quote) in a padded simdjson buffer, and len
+     * must be the number of raw key bytes (the distance to the closing quote).
+     * Returns the selector index in [0, N) on match, or N on miss.
+     *
+     * Prefer this overload when the caller can obtain the key length cheaply (for
+     * example, object::for_each derives it from the structural index rather than
+     * re-scanning for the closing quote).
+     */
+    static simdjson_really_inline std::size_t match_raw(const char* p, std::size_t len) noexcept {
+        if (len == 0 || len > max_key_len) { return N; }
+
+        if constexpr (window.ok) {
+            // One 8-bit window selects the only possible candidate key;
+            // match_window_candidate confirms it (bytes + closing quote). p sits
+            // in a padded buffer and the window stays within the shortest key +
+            // quote, so the two-byte read is always in bounds. len is unused here
+            // because the quote check already pins the key's end.
+            std::uint8_t ki = window.window_to_key[
+                key_selector_detail::read_window(p, window.byte_offset, window.shift)];
+            if (ki >= N) { return N; }
+            return key_selector_detail::match_window_candidate(
+                p, ki, window, std::make_index_sequence<N>{});
+        }
+
+        std::size_t slot;
+        if (phf.num_positions == key_selector_detail::HD_MODE) {
+            // Hash-and-Displace: bucket displacement + per-key hash.
+            std::string_view key(p, len);
+            std::size_t bucket = key_selector_detail::hd_bucket_hash(key);
+            std::size_t kh = (phf.hd_hash_variant == key_selector_detail::HD_HASH_2BYTE_FLAG)
+                ? key_selector_detail::hd_key_hash_2(key)
+                : key_selector_detail::hd_key_hash_4(key);
+            slot = (phf.asso_values[0][bucket] + kh) & (table_size - 1);
+        } else {
+            // gperf: h = len + sum of asso_values over the selected positions.
+            // positions / num_positions / asso_values are compile-time constants,
+            // so this loop fully unrolls. The idx < len guard mirrors the
+            // generator's char_at()-> 256 -> skip behavior for out-of-range
+            // positions (required: arbitrary positions may exceed a key's length).
+            std::size_t h = len;
+            for (std::uint8_t i = 0; i < phf.num_positions; ++i) {
+                std::uint8_t pos = phf.positions[i];
+                std::size_t idx = (pos == key_selector_detail::POS_LAST_CHAR)
+                                  ? (len - std::size_t{1})
+                                  : static_cast<std::size_t>(pos);
+                if (idx < len) {
+                    h += phf.asso_values[i][static_cast<unsigned char>(p[idx])];
+                }
+            }
+            slot = h & (table_size - 1);
+        }
+
+        std::uint8_t ki = phf.slot_to_key[slot];
+        if (ki >= N) { return N; }
+        if (phf.slot_key_len[slot] != len) { return N; }
+        if (!key_selector_detail::compare_key_bytes<max_key_len>(
+                p, phf.slot_key_bytes[slot].data(), len)) { return N; }
+        return ki;
+    }
+
+    /**
+     * Look up a JSON key. rjs must point just after an opening quote in a padded
+     * simdjson buffer. Returns the selector index in [0, N) on match, or N on miss.
+     * The key length is recovered with a SIMD scan for the closing quote; callers
+     * that already know the length should use the (p, len) overload above.
+     */
+    static simdjson_really_inline std::size_t match_raw(raw_json_string rjs) noexcept {
+        const char* p = rjs.raw();
+        if constexpr (window.ok) {
+            // One 8-bit window picks the candidate; verifying the candidate's
+            // bytes and its closing '"' confirms the full key, so the length scan
+            // is unnecessary. The window read is in bounds (padding), and the
+            // candidate length is at most max_key_len.
+            std::uint8_t ki = window.window_to_key[
+                key_selector_detail::read_window(p, window.byte_offset, window.shift)];
+            if (ki >= N) { return N; }
+            return key_selector_detail::match_window_candidate(
+                p, ki, window, std::make_index_sequence<N>{});
+        }
+        return match_raw(p, key_selector_detail::scan_key_length<max_key_len>(p));
+    }
+
+    /** Return the key text at selector index i (i in [0, N)). */
+    static constexpr std::string_view key_at(std::size_t i) noexcept {
+        return keys[i];
+    }
+
+    /**
+     * Return a complete, human-readable, multi-line description of how this
+     * selector classifies a key: which algorithm was selected at compile time
+     * (single 8-bit window, gperf-style perfect hash, or hash-and-displace), the
+     * exact bytes/positions it inspects, and the contents of the lookup tables
+     * (which window bytes or hash slots map to which key). The text mirrors what
+     * match_raw() does step by step.
+     *
+     * Everything it reports is derived from the compile-time tables, so describe()
+     * is itself usable in a constant expression when the standard library supports
+     * constexpr std::string (__cpp_lib_constexpr_string):
+     *
+     *   static_assert(!key_selector<"name", "city">::describe().empty());
+     *
+     * It allocates a std::string and is meant for documentation, debugging and
+     * tests, not for any hot path.
+     */
+    static SIMDJSON_CONSTEXPR_STRING std::string describe() {
+        std::string s;
+        s += "key_selector: ";
+        key_selector_detail::append_uint(s, N);
+        s += " keys, max key length ";
+        key_selector_detail::append_uint(s, max_key_len);
+        s += "\nkeys:\n";
+        for (std::size_t i = 0; i < N; ++i) {
+            s += "  [";
+            key_selector_detail::append_uint(s, i);
+            s += "] \"";
+            s += keys[i];
+            s += "\" (length ";
+            key_selector_detail::append_uint(s, keys[i].size());
+            s += ")\n";
+        }
+        if constexpr (window.ok) {
+            // Mirrors the window fast path of match_raw().
+            s += "algorithm: single 8-bit window\n";
+            s += "  step 1: read 2 bytes at offset ";
+            key_selector_detail::append_uint(s, window.byte_offset);
+            s += ", interpret them as a little-endian 16-bit value, shift right by ";
+            key_selector_detail::append_uint(s, window.shift);
+            s += " bits, and keep the low 8 bits\n";
+            s += "  step 2: map that byte through a 256-entry table to a key index (";
+            key_selector_detail::append_uint(s, N);
+            s += " means no match):\n";
+            for (std::size_t b = 0; b < 256; ++b) {
+                if (window.window_to_key[b] < N) {
+                    s += "    byte ";
+                    key_selector_detail::append_byte(s, static_cast<unsigned>(b));
+                    s += " -> key ";
+                    key_selector_detail::append_uint(s, window.window_to_key[b]);
+                    s += "\n";
+                }
+            }
+            s += "  step 3: confirm the candidate by checking the closing quote sits at the key's length and comparing the key bytes\n";
+        } else {
+            // Mirrors the perfect-hash path of match_raw().
+            if constexpr (phf.num_positions == key_selector_detail::HD_MODE) {
+                s += "algorithm: hash-and-displace perfect hash\n";
+                s += "  step 1: bucket = (first_byte + last_byte*3 + length*17) mod 256\n";
+                s += "  step 2: keyhash = base-31 rolling hash of the length and the first ";
+                key_selector_detail::append_uint(s, phf.hd_hash_variant);
+                s += " bytes\n";
+                s += "  step 3: slot = (displacement[bucket] + keyhash) mod ";
+                key_selector_detail::append_uint(s, table_size);
+                s += "\n  step 4: slot_to_key[slot] gives the candidate key index\n";
+                s += "  per-key derivation:\n";
+                for (std::size_t i = 0; i < N; ++i) {
+                    std::string_view k = keys[i];
+                    std::size_t bucket = key_selector_detail::hd_bucket_hash(k);
+                    std::size_t kh = (phf.hd_hash_variant == key_selector_detail::HD_HASH_2BYTE_FLAG)
+                        ? key_selector_detail::hd_key_hash_2(k)
+                        : key_selector_detail::hd_key_hash_4(k);
+                    std::size_t slot = (phf.asso_values[0][bucket] + kh) & (table_size - 1);
+                    s += "    \"";
+                    s += k;
+                    s += "\": bucket=";
+                    key_selector_detail::append_uint(s, bucket);
+                    s += " displacement=";
+                    key_selector_detail::append_uint(s, phf.asso_values[0][bucket]);
+                    s += " keyhash=";
+                    key_selector_detail::append_uint(s, kh);
+                    s += " slot=";
+                    key_selector_detail::append_uint(s, slot);
+                    s += "\n";
+                }
+            } else {
+                s += "algorithm: gperf-style perfect hash over ";
+                key_selector_detail::append_uint(s, phf.num_positions);
+                s += " character position(s)\n";
+                s += "  step 1: h = key length\n";
+                s += "  step 2: for each position below, add its association value for the key byte there (a position past the key's length contributes 0):\n";
+                for (std::size_t i = 0; i < phf.num_positions; ++i) {
+                    s += "    position ";
+                    if (phf.positions[i] == key_selector_detail::POS_LAST_CHAR) {
+                        s += "last character";
+                    } else {
+                        s += "byte index ";
+                        key_selector_detail::append_uint(s, phf.positions[i]);
+                    }
+                    s += "\n";
+                }
+                s += "  step 3: slot = h mod ";
+                key_selector_detail::append_uint(s, table_size);
+                s += " (a power of two, applied as a bitmask)\n";
+                s += "  step 4: slot_to_key[slot] gives the candidate key index\n";
+                s += "  per-key derivation:\n";
+                for (std::size_t i = 0; i < N; ++i) {
+                    std::string_view k = keys[i];
+                    std::size_t h = k.size();
+                    for (std::size_t pi = 0; pi < phf.num_positions; ++pi) {
+                        std::size_t pos = phf.positions[pi];
+                        std::size_t idx = (pos == key_selector_detail::POS_LAST_CHAR)
+                                          ? (k.size() - 1) : pos;
+                        if (idx < k.size()) {
+                            h += phf.asso_values[pi][static_cast<unsigned char>(k[idx])];
+                        }
+                    }
+                    std::size_t slot = h & (table_size - 1);
+                    s += "    \"";
+                    s += k;
+                    s += "\": h=";
+                    key_selector_detail::append_uint(s, h);
+                    s += " slot=";
+                    key_selector_detail::append_uint(s, slot);
+                    s += "\n";
+                }
+            }
+            s += "  occupied slots (slot -> key):\n";
+            for (std::size_t slot = 0; slot < table_size; ++slot) {
+                if (phf.slot_to_key[slot] < N) {
+                    s += "    slot ";
+                    key_selector_detail::append_uint(s, slot);
+                    s += " -> key ";
+                    key_selector_detail::append_uint(s, phf.slot_to_key[slot]);
+                    s += " (\"";
+                    s += keys[phf.slot_to_key[slot]];
+                    s += "\", length ";
+                    key_selector_detail::append_uint(s, phf.slot_key_len[slot]);
+                    s += ")\n";
+                }
+            }
+            s += "  confirm the candidate by checking the key length matches and comparing the key bytes\n";
+        }
+        return s;
+    }
+};
+
+namespace key_selector_detail {
+template <typename> struct is_key_selector : std::false_type {};
+template <constevalutil::fixed_string... Keys>
+struct is_key_selector<key_selector<Keys...>> : std::true_type {};
+} // namespace key_selector_detail
+
+/**
+ * Matches any instantiation of key_selector<Keys...>. Used to constrain
+ * object::for_each so that passing a non-selector type yields a clear
+ * constraint error rather than a cascade of failures inside for_each.
+ */
+template <typename T>
+concept key_selector_type = key_selector_detail::is_key_selector<T>::value;
+
+} // namespace ondemand
+} // namespace arm64
+} // namespace simdjson
+
+#endif // SIMDJSON_SUPPORTS_CONCEPTS
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+/* end file simdjson/generic/ondemand/key_selector.h for arm64 */
 /* including simdjson/generic/ondemand/object.h for arm64: #include "simdjson/generic/ondemand/object.h" */
 /* begin file simdjson/generic/ondemand/object.h for arm64 */
 #ifndef SIMDJSON_GENERIC_ONDEMAND_OBJECT_H
@@ -67386,6 +83268,7 @@ public:
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/implementation_simdjson_result_base.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/key_selector.h" */
 /* amalgamation skipped (editor-only): #include <vector> */
 /* amalgamation skipped (editor-only): #if SIMDJSON_STATIC_REFLECTION && SIMDJSON_SUPPORTS_CONCEPTS */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_string_builder.h"  // for constevalutil::fixed_string */
@@ -67396,6 +83279,114 @@ namespace simdjson {
 namespace arm64 {
 namespace ondemand {

+#if SIMDJSON_SUPPORTS_CONCEPTS
+/**
+ * Result of object::for_each: the first error encountered (SUCCESS if none) and
+ * the number of distinct selector keys that matched during the walk. A
+ * matched_count equal to Selector::size() means every selected key was present
+ * in the object. Implicitly converts to error_code so existing callers that only
+ * care about the error (including SIMDJSON_TRY and the test ASSERT_* macros) keep
+ * working unchanged.
+ *
+ * On error paths (a handler returns a non-SUCCESS error_code, or a direct-target
+ * deserialization via value::get fails), matched_count reflects only the number
+ * of distinct selected keys that were successfully handled *before* the failure;
+ * the failing key is not counted. This is the current behavior.
+ *
+ * Marked [[nodiscard]]: for_each surfaces parse/type errors (from a matched
+ * value, a returning handler, or the structural walk itself) only through this
+ * result, so it must not be silently dropped. This matters most in builds
+ * without exceptions, where it is the only channel for those errors. To
+ * deliberately ignore it, assign to a variable or cast to void.
+ */
+struct [[nodiscard]] for_each_result {
+  error_code error{SUCCESS};
+  std::size_t matched_count{0};
+  constexpr operator error_code() const noexcept { return error; }
+};
+
+namespace key_selector_for_each_detail {
+/**
+ * A per-key handler for the variadic object::for_each is either:
+ *   - an invocable taking a value (run custom logic for that field), or
+ *   - a deserialization target T, in which case the matched value is assigned
+ *     directly via value::get(T&).
+ * The target form lets callers bind fields straight to variables without a
+ * lambda per key -- e.g. obj.for_each<"name","city","age">(name, city, age) --
+ * while still allowing a lambda in any position when custom logic is needed
+ * (for example to descend into a nested object).
+ */
+template <typename H>
+concept field_handler =
+    std::is_invocable_v<std::remove_reference_t<H>&, value> ||
+    ::simdjson::deserializable<std::remove_cvref_t<H>, value>;
+
+/**
+ * noexcept-ness of handling a single handler: nothrow-invocability for the
+ * callback form, or nothrow-deserializability for the direct-target form
+ * (builtin scalar/string targets are noexcept; custom targets follow their
+ * own tag_invoke noexcept specification).
+ */
+template <typename H>
+inline constexpr bool nothrow_field_handler_v =
+    std::is_invocable_v<std::remove_reference_t<H>&, value>
+        ? std::is_nothrow_invocable_v<std::remove_reference_t<H>&, value>
+        : ::simdjson::nothrow_deserializable<std::remove_cvref_t<H>, value>;
+
+/**
+ * nothrow_field_handler_v folded over a whole handler pack. The variadic
+ * for_each overloads spell their conditional-noexcept specifier with this
+ * single id-expression rather than an inline fold expression. MSVC (VS18)
+ * mishandles a fold-expression that appears directly in the noexcept-specifier
+ * of an out-of-line template definition: it fails to recognize the definition's
+ * specifier as matching the in-class declaration's and reports a spurious
+ * C2382 ("redefinition; different exception specifications"). Naming the fold
+ * here keeps the specifier a plain identifier, which matches reliably.
+ */
+template <typename... Handlers>
+inline constexpr bool nothrow_field_handlers_v =
+    (nothrow_field_handler_v<Handlers> && ...);
+} // namespace key_selector_for_each_detail
+#endif
+
+/**
+ * An opaque snapshot of an object's scanning position, obtained from
+ * object::get_current_position() and consumed by object::revert_position().
+ *
+ * This bundles the raw token position with the iteration depth that was
+ * live at the moment of capture: a scalar field value (e.g. a number)
+ * unwinds back to the object's own depth once fully consumed, while a
+ * compound value (array/object) not yet fully consumed stays one level
+ * deeper. The correct depth to restore to therefore depends on what was
+ * live when the snapshot was taken, not on the object itself, which is
+ * why both pieces travel together here rather than being recomputed later.
+ *
+ * The members are private and only constructible by object itself: a
+ * hand-built value here would let revert_position() jump to an arbitrary,
+ * unvalidated position and depth, and this type carries no way to check
+ * that a given snapshot still refers to a live object. Only ever pass
+ * along a value you got from get_current_position(), and only back to the
+ * same object, before that object has been reset() or the parser has
+ * iterate()d a new document: neither invalidates a snapshot in a way this
+ * type can detect.
+ */
+struct object_position {
+  /**
+   * Default-constructed so a variable can be declared and assigned later,
+   * matching e.g. document()/object(). Not a valid position to revert to.
+   */
+  simdjson_inline object_position() noexcept = default;
+
+private:
+  token_position position{};
+  depth_t depth{};
+
+  simdjson_inline object_position(token_position position_, depth_t depth_) noexcept
+    : position(position_), depth(depth_) {}
+
+  friend class object;
+};
+
 /**
  * A forward-only JSON object field iterator.
  */
@@ -67414,8 +83405,19 @@ public:
    * Using the iterator directly is also possible but error-prone and discouraged. In particular,
    * you must dereference the iterator exactly once per iteration (before calling '++').
    * Doing otherwise is unsafe and may lead to errors. You are responsible for ensuring
+   *
+   * With SIMDJSON_DEVELOPMENT_CHECKS, the iterator locks this object while it is
+   * alive, so that reentrant access (find_field(), reset(), ...) is reported as
+   * OUT_OF_ORDER_ITERATION.
+   */
+  simdjson_inline simdjson_result<object_iterator> begin() & noexcept;
+  /**
+   * Get an iterator to the start of a temporary object, e.g., `v.get_object().begin()`.
+   *
+   * The iterator does not depend on the object instance and may outlive it, so
+   * it does not lock it.
    */
-  simdjson_inline simdjson_result<object_iterator> begin() noexcept;
+  simdjson_inline simdjson_result<object_iterator> begin() && noexcept;
   simdjson_inline simdjson_result<object_iterator> end() noexcept;
   /**
    * Look up a field by name on an object (order-sensitive). By order-sensitive, we mean that
@@ -67427,10 +83429,11 @@ public:
    *
    * ```cpp
    * simdjson::ondemand::parser parser;
-   * auto obj = parser.parse(R"( { "x": 1, "y": 2, "z": 3 } )"_padded);
-   * double z = obj.find_field("z");
-   * double y = obj.find_field("y");
-   * double x = obj.find_field("x");
+   * auto json = R"( { "x": 1, "y": 2, "z": 3 } )"_padded;
+   * auto doc = parser.iterate(json);
+   * double z = doc.find_field("z");
+   * double y = doc.find_field("y");
+   * double x = doc.find_field("x");
    * ```
    * If you have multiple fields with a matching key ({"x": 1,  "x": 1}) be mindful
    * that only one field is returned.
@@ -67503,6 +83506,100 @@ public:
   /** @overload simdjson_inline simdjson_result<value> find_field_unordered(std::string_view key) & noexcept; */
   simdjson_inline simdjson_result<value> operator[](std::string_view key) && noexcept;

+#if SIMDJSON_SUPPORTS_CONCEPTS
+  /**
+   * Walk this object once and invoke on_match(selector_index, value) for each
+   * field whose key is in the compile-time key_selector Selector, in JSON order
+   * (first occurrence of a duplicate key wins). Iteration stops once all
+   * Selector::size() keys have matched or the object ends. The value is consumed
+   * in place, so this is a low-overhead way to extract a known set of fields
+   * regardless of their order in the JSON.
+   *
+   * Like other object iteration in simdjson, for_each consumes the object by
+   * advancing the underlying iterator state; after the call the same object
+   * instance should not be used for further field access or iteration.
+   *
+   * Usage:
+   *   using sel_t = ondemand::key_selector<"id", "text", "user">;
+   *   obj.for_each<sel_t>([&](std::size_t i, ondemand::value v) {
+   *     switch (i) { case 0: ...; case 1: ...; }
+   *   });
+   *
+   * Limitations (see key_selector): each key must be at most 63 characters long,
+   * and the number of keys should be moderate (hard limit 255; a handful is
+   * best, as the compile-time perfect hash may fail or slow compilation for
+   * large key sets). The keys must be distinct, non-empty, and free of backslash, double-quote and
+   * null bytes.
+   *
+   * The callback may return either void or an error_code. When it returns an
+   * error_code, the walk stops at the first non-SUCCESS result and that error is
+   * returned, which lets the callback surface value-parse errors.
+   *
+   * This function is conditionally noexcept: it is noexcept exactly when invoking
+   * the callback is noexcept. The callback runs inside this frame, so a throwing
+   * callback (e.g. one using the exception-throwing conversions like
+   * std::string_view(value) or uint64_t(value)) makes for_each potentially
+   * throwing too -- the exception propagates to the caller instead of crossing a
+   * noexcept boundary and calling std::terminate.
+   *
+   * @returns a for_each_result holding the first error encountered while walking
+   *          the object (including any error returned by the callback, SUCCESS if
+   *          none) and the number of distinct selector keys that matched. The
+   *          result converts implicitly to error_code, so callers that only need
+   *          the error can ignore the count.
+   */
+  template <typename Selector, typename Func>
+    requires key_selector_type<Selector> &&
+             std::is_invocable_v<Func&, std::size_t, value>
+  simdjson_inline for_each_result for_each(Func&& on_match)
+      noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, value>);
+
+  /**
+   * Variadic per-key form. Provide exactly one handler per key in the Selector
+   * (compiler-enforced). Handlers are processed in JSON document order for the
+   * matching keys. Each handler is either:
+   *   - a deserialization target (a variable), in which case the matched value
+   *     is assigned to it via value::get -- no lambda required; or
+   *   - an invocable taking the ondemand::value (for custom logic such as
+   *     descending into a nested object). It may return void or error_code;
+   *     returning error_code lets you surface parse/type errors.
+   * The two styles may be mixed freely, one handler per key.
+   *
+   * Example (bind fields straight to variables):
+   *   using fields = ondemand::key_selector<"name", "city", "age">;
+   *   obj.for_each<fields>(name, city, age);
+   *
+   * Example (mixing a target and a lambda):
+   *   obj.for_each<ondemand::key_selector<"id", "user">>(
+   *     id,                                          // assigned via value::get
+   *     [&](ondemand::value v){ u = read_user(v); }  // custom logic
+   *   );
+   *
+   * The index-based single-callback form (taking (size_t, value)) remains
+   * available for shared-state or more complex per-key logic.
+   */
+  template <typename Selector, typename... Handlers>
+    requires key_selector_type<Selector> &&
+             (sizeof...(Handlers) == Selector::size()) &&
+             (key_selector_for_each_detail::field_handler<Handlers> && ...)
+  simdjson_inline for_each_result for_each(Handlers&&... on_match)
+      noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
+  /**
+   * Direct-key shorthand. Equivalent to for_each<key_selector<Keys...>>(...).
+   * Lets you write the keys inline without a separate using/alias, binding each
+   * field straight to a variable (or a lambda, see the Selector form above):
+   *
+   *   obj.for_each<"name", "city", "age">(name, city, age);
+   */
+  template <constevalutil::fixed_string... Keys, typename... Handlers>
+    requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+             (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+             (key_selector_for_each_detail::field_handler<Handlers> && ...)
+  simdjson_inline for_each_result for_each(Handlers&&... on_match)
+      noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+#endif
+
   /**
    * Get the value associated with the given JSON pointer. We use the RFC 6901
    * https://tools.ietf.org/html/rfc6901 standard, interpreting the current node
@@ -67579,6 +83676,34 @@ public:
    * @returns true if the object contains some elements (not empty)
    */
   inline simdjson_result<bool> reset() & noexcept;
+  /**
+   * Get an opaque token representing the object's current scanning position.
+   * Pass it to revert_position() to return to this exact point later, without
+   * paying the cost of a full reset() and re-scan from the beginning.
+   *
+   * A typical use is an optional field that may or may not be next: capture
+   * the position, attempt find_field(), and on NO_SUCH_FIELD, revert_position()
+   * instead of reset() so that fields already consumed are not rescanned.
+   *
+   * The returned token is only valid for this object, and only until it is
+   * reset() or the parser iterate()s a new document; using it after either
+   * is undefined behavior (see object_position).
+   *
+   * @returns An opaque position token.
+   */
+  simdjson_inline object_position get_current_position() const noexcept;
+  /**
+   * Return the object's scanning position to a snapshot previously obtained
+   * from get_current_position(). Unlike reset(), this does not rescan the
+   * object from the beginning: fields before the captured position remain
+   * consumed, and scanning resumes exactly where the snapshot was captured.
+   *
+   * @param position A snapshot previously returned by get_current_position(),
+   *        for this same object.
+   * @returns SUCCESS, or OUT_OF_ORDER_ITERATION if called during active
+   *          iteration (SIMDJSON_DEVELOPMENT_CHECKS builds only).
+   */
+  simdjson_inline error_code revert_position(object_position position) noexcept;
   /**
    * This method scans the beginning of the object and checks whether the
    * object is empty.
@@ -67624,7 +83749,7 @@ public:
    */
   template <typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out)
-     noexcept(custom_deserializable<T, object> ? nothrow_custom_deserializable<T, object> : true) {
+     noexcept(nothrow_gettable<T, object>) {
     static_assert(custom_deserializable<T, object>);
     return deserialize(*this, out);
   }
@@ -67636,7 +83761,7 @@ public:
    */
   template <typename T>
   simdjson_inline simdjson_result<T> get()
-    noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+    noexcept(nothrow_gettable<T, object>)
   {
     static_assert(std::is_default_constructible<T>::value, "The specified type is not default constructible.");
     T out{};
@@ -67688,10 +83813,18 @@ protected:
   simdjson_warn_unused simdjson_inline error_code find_field_raw(const std::string_view key) noexcept;

   value_iterator iter{};
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  bool locked{false};
+  simdjson_inline void set_locked(bool _locked) noexcept;
+#endif

   friend class value;
   friend class document;
   friend struct simdjson_result<object>;
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  friend class object_iterator;
+  friend struct simdjson_result<object_iterator>;
+#endif
 };

 } // namespace ondemand
@@ -67707,7 +83840,8 @@ public:
   simdjson_inline simdjson_result(error_code error) noexcept; ///< @private
   simdjson_inline simdjson_result() noexcept = default;

-  simdjson_inline simdjson_result<arm64::ondemand::object_iterator> begin() noexcept;
+  simdjson_inline simdjson_result<arm64::ondemand::object_iterator> begin() & noexcept;
+  simdjson_inline simdjson_result<arm64::ondemand::object_iterator> begin() && noexcept;
   simdjson_inline simdjson_result<arm64::ondemand::object_iterator> end() noexcept;
   simdjson_inline simdjson_result<arm64::ondemand::value> find_field(std::string_view key) & noexcept;
   simdjson_inline simdjson_result<arm64::ondemand::value> find_field(std::string_view key) && noexcept;
@@ -67725,6 +83859,8 @@ public:
 #endif
   simdjson_inline error_code for_each_at_path_with_wildcard(std::string_view json_path, Func&& callback) noexcept;
   inline simdjson_result<bool> reset() noexcept;
+  inline simdjson_result<arm64::ondemand::object_position> get_current_position() noexcept;
+  inline error_code revert_position(arm64::ondemand::object_position position) noexcept;
   inline simdjson_result<bool> is_empty() noexcept;
   inline simdjson_result<size_t> count_fields() & noexcept;
   inline simdjson_result<std::string_view> raw_json() noexcept;
@@ -67732,7 +83868,7 @@ public:
   // TODO: move this code into object-inl.h

   template<typename T>
-  simdjson_inline simdjson_result<T> get() noexcept {
+  simdjson_inline simdjson_result<T> get() noexcept(nothrow_gettable<T, arm64::ondemand::object>) {
     if (error()) { return error(); }
     if constexpr (std::is_same_v<T, arm64::ondemand::object>) {
       return first;
@@ -67740,7 +83876,7 @@ public:
     return first.get<T>();
   }
   template<typename T>
-  simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept {
+  simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept(nothrow_gettable<T, arm64::ondemand::object>) {
     if (error()) { return error(); }
     if constexpr (std::is_same_v<T, arm64::ondemand::object>) {
       out = first;
@@ -67750,6 +83886,39 @@ public:
     return SUCCESS;
   }

+  /**
+   * Forwards to object::for_each on the underlying object, so error-code-style
+   * chains (e.g. doc["x"].get_object()) can call for_each without first
+   * extracting the object. If this result holds an error, that error is returned
+   * (with a zero match count) and the callback is not invoked. See
+   * object::for_each for the semantics.
+   */
+  template <typename Selector, typename Func>
+    requires arm64::ondemand::key_selector_type<Selector> &&
+             std::is_invocable_v<Func&, std::size_t, arm64::ondemand::value>
+  simdjson_inline arm64::ondemand::for_each_result for_each(Func&& on_match)
+      noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, arm64::ondemand::value>);
+
+  /**
+   * Forwarding overload for the variadic per-key form (targets and/or lambdas).
+   */
+  template <typename Selector, typename... Handlers>
+    requires arm64::ondemand::key_selector_type<Selector> &&
+             (sizeof...(Handlers) == Selector::size()) &&
+             (arm64::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+  simdjson_inline arm64::ondemand::for_each_result for_each(Handlers&&... on_match)
+      noexcept(arm64::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
+  /**
+   * Forwarding overload for the direct-key variadic form.
+   */
+  template <constevalutil::fixed_string... Keys, typename... Handlers>
+    requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+             (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+             (arm64::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+  simdjson_inline arm64::ondemand::for_each_result for_each(Handlers&&... on_match)
+      noexcept(arm64::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
 #if SIMDJSON_STATIC_REFLECTION
   // TODO: move this code into object-inl.h
   template<constevalutil::fixed_string... FieldNames, typename T>
@@ -67790,6 +83959,15 @@ public:
    */
   simdjson_inline object_iterator() noexcept = default;

+#if SIMDJSON_DEVELOPMENT_CHECKS
+   simdjson_inline ~object_iterator() noexcept;
+
+   simdjson_inline object_iterator(object_iterator&&) noexcept;
+   simdjson_inline object_iterator& operator=(object_iterator&&) noexcept;
+   simdjson_inline object_iterator(const object_iterator&) noexcept;
+   simdjson_inline object_iterator& operator=(const object_iterator&) noexcept;
+#endif
+
   //
   // Iterator interface
   //
@@ -67809,6 +83987,9 @@ public:
 private:
 #if SIMDJSON_DEVELOPMENT_CHECKS
    bool has_been_referenced{false};
+   object* parent{nullptr};
+
+   simdjson_inline object_iterator(const value_iterator &_iter, object* _parent) noexcept;
 #endif
   /**
    * The underlying JSON iterator.
@@ -67854,6 +84035,191 @@ public:

 #endif // SIMDJSON_GENERIC_ONDEMAND_OBJECT_ITERATOR_H
 /* end file simdjson/generic/ondemand/object_iterator.h for arm64 */
+/* including simdjson/generic/ondemand/ranges.h for arm64: #include "simdjson/generic/ondemand/ranges.h" */
+/* begin file simdjson/generic/ondemand/ranges.h for arm64 */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_RANGES_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_ONDEMAND_RANGES_H */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/field.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#if SIMDJSON_SUPPORTS_RANGES
+
+namespace simdjson {
+namespace arm64 {
+namespace ondemand {
+
+/**
+ * A ranges-compatible iterator adapter for JSON arrays.
+ *
+ * Wraps array_iterator to satisfy std::input_iterator by providing:
+ * - const operator* (via mutable internal state)
+ * - post-increment operator
+ * - iterator_concept tag
+ *
+ * The mutable approach is standard for single-pass input iterators that
+ * read from external sources (similar to std::istream_iterator).
+ */
+class array_range_iterator {
+public:
+  using iterator_concept = std::input_iterator_tag;
+  using value_type = simdjson_result<value>;
+  using reference = simdjson_result<value>;
+  using difference_type = std::ptrdiff_t;
+
+  simdjson_inline array_range_iterator() noexcept = default;
+  simdjson_inline explicit array_range_iterator(array_iterator iter) noexcept;
+
+  /**
+   * Get the current element. Const-qualified for std::indirectly_readable;
+   * internally delegates to the mutable wrapped iterator.
+   */
+  simdjson_inline simdjson_result<value> operator*() const noexcept;
+  simdjson_inline array_range_iterator& operator++() noexcept;
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+  simdjson_inline void operator++(int) noexcept;
+SIMDJSON_POP_DISABLE_WARNINGS
+  /**
+   * Comparison delegates to array_iterator::operator==, which checks
+   * whether the underlying parser has finished the array (depth-based).
+   */
+  simdjson_inline friend bool operator==(const array_range_iterator& a,
+                                         const array_range_iterator& b) noexcept {
+    return a.iter_ == b.iter_;
+  }
+
+private:
+  mutable array_iterator iter_{};
+};
+
+/**
+ * A std::ranges::view over a JSON array.
+ *
+ * Wraps an ondemand::array and exposes begin()/end() that return
+ * array_range_iterator (satisfying std::input_iterator), enabling
+ * use with std::views::transform and other range adaptors.
+ *
+ * If the array's begin() returns an error (only possible under
+ * SIMDJSON_DEVELOPMENT_CHECKS), the range will be empty and error()
+ * will return the error code.
+ *
+ * Usage:
+ *   ondemand::parser parser;
+ *   auto doc = parser.iterate(json);
+ *   auto arr = doc.get_array().value();
+ *   for (auto elem : ondemand::get_range(arr)) { ... }
+ */
+class array_range {
+public:
+  simdjson_inline array_range() noexcept = default;
+  simdjson_inline explicit array_range(array& arr) noexcept;
+
+  simdjson_inline array_range_iterator begin() noexcept;
+  simdjson_inline array_range_iterator end() noexcept;
+
+  /** Returns SUCCESS if the range was created successfully, or the error code otherwise. */
+  simdjson_inline error_code error() const noexcept { return error_; }
+
+private:
+  array_iterator begin_{};
+  array_iterator end_{};
+  error_code error_{SUCCESS};
+};
+
+/**
+ * A ranges-compatible iterator adapter for JSON objects.
+ *
+ * Wraps object_iterator to satisfy std::input_iterator, yielding
+ * simdjson_result<field> elements (key-value pairs).
+ */
+class object_range_iterator {
+public:
+  using iterator_concept = std::input_iterator_tag;
+  using value_type = simdjson_result<field>;
+  using reference = simdjson_result<field>;
+  using difference_type = std::ptrdiff_t;
+
+  simdjson_inline object_range_iterator() noexcept = default;
+  simdjson_inline explicit object_range_iterator(object_iterator iter) noexcept;
+
+  simdjson_inline simdjson_result<field> operator*() const noexcept;
+  simdjson_inline object_range_iterator& operator++() noexcept;
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+  simdjson_inline void operator++(int) noexcept;
+SIMDJSON_POP_DISABLE_WARNINGS
+  simdjson_inline friend bool operator==(const object_range_iterator& a,
+                                         const object_range_iterator& b) noexcept {
+    return a.iter_ == b.iter_;
+  }
+
+private:
+  mutable object_iterator iter_{};
+};
+
+/**
+ * A std::ranges::view over a JSON object.
+ *
+ * Wraps an ondemand::object and exposes begin()/end() that return
+ * object_range_iterator, enabling use with range adaptors.
+ *
+ * If the object's begin() returns an error, the range will be empty
+ * and error() will return the error code.
+ */
+class object_range {
+public:
+  simdjson_inline object_range() noexcept = default;
+  simdjson_inline explicit object_range(object& obj) noexcept;
+
+  simdjson_inline object_range_iterator begin() noexcept;
+  simdjson_inline object_range_iterator end() noexcept;
+
+  /** Returns SUCCESS if the range was created successfully, or the error code otherwise. */
+  simdjson_inline error_code error() const noexcept { return error_; }
+
+private:
+  object_iterator begin_{};
+  object_iterator end_{};
+  error_code error_{SUCCESS};
+};
+
+/** Get a std::ranges compatible view over a JSON array. */
+simdjson_inline array_range get_range(array& arr) noexcept;
+
+/** Get a std::ranges compatible view over a JSON object (key-value pairs). */
+simdjson_inline object_range get_key_value_range(object& obj) noexcept;
+
+#if SIMDJSON_EXCEPTIONS
+/** Get a std::ranges compatible view, unwrapping the simdjson_result (throws on error). */
+simdjson_inline array_range get_range(simdjson_result<array> result);
+
+/** Get a std::ranges compatible view, unwrapping the simdjson_result (throws on error). */
+simdjson_inline object_range get_key_value_range(simdjson_result<object> result);
+#endif // SIMDJSON_EXCEPTIONS
+
+} // namespace ondemand
+} // namespace arm64
+} // namespace simdjson
+
+namespace std {
+namespace ranges {
+template<>
+inline constexpr bool enable_view<simdjson::arm64::ondemand::array_range> = true;
+template<>
+inline constexpr bool enable_view<simdjson::arm64::ondemand::object_range> = true;
+} // namespace ranges
+} // namespace std
+
+#endif // SIMDJSON_SUPPORTS_RANGES
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_RANGES_H
+/* end file simdjson/generic/ondemand/ranges.h for arm64 */
 /* including simdjson/generic/ondemand/serialization.h for arm64: #include "simdjson/generic/ondemand/serialization.h" */
 /* begin file simdjson/generic/ondemand/serialization.h for arm64 */
 #ifndef SIMDJSON_GENERIC_ONDEMAND_SERIALIZATION_H
@@ -67986,12 +84352,14 @@ inline std::ostream& operator<<(std::ostream& out, simdjson::simdjson_result<sim
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/annotations.h" */
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */

 #include <concepts>
 #include <limits>
 #if SIMDJSON_STATIC_REFLECTION
 #include <meta>
+#include <vector>
 // #include <static_reflection> // for std::define_static_string - header not available yet
 #endif

@@ -68016,10 +84384,25 @@ error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept {

 template <std::floating_point T>
 error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept {
-  double x;
-  SIMDJSON_TRY(val.get_double().get(x));
-  out = static_cast<T>(x);
-  return SUCCESS;
+  if constexpr (std::is_same_v<T, float>) {
+    // Going through binary64 and then rounding to binary32 would round twice
+    // and could produce a value that is not the float nearest to the JSON
+    // number, so we parse to binary32 directly.
+    return val.get_float().get(out);
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  } else if constexpr (std::is_same_v<T, std::float32_t>) {
+    // Same reason as float.
+    float x;
+    SIMDJSON_TRY(val.get_float().get(x));
+    out = static_cast<T>(x);
+    return SUCCESS;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+  } else {
+    double x;
+    SIMDJSON_TRY(val.get_double().get(x));
+    out = static_cast<T>(x);
+    return SUCCESS;
+  }
 }

 template <std::signed_integral T>
@@ -68055,11 +84438,59 @@ template <concepts::constructible_from_string_view T, typename ValT>
 error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(std::is_nothrow_constructible_v<T, std::string_view>) {
   std::string_view str;
   SIMDJSON_TRY(val.get_string().get(str));
-  out = T{str};
+  if constexpr (requires { out.assign(str.data(), str.size()); }) {
+    // Copy straight into out (e.g., std::string): building a temporary and
+    // move-assigning it is markedly slower.
+    out.assign(str.data(), str.size());
+  } else {
+    out = T{str};
+  }
+  return SUCCESS;
+}
+
+#if SIMDJSON_SUPPORTS_CHAR8_T
+// any C++20 char8_t string-like type (can be constructed from std::u8string_view),
+// such as std::u8string
+template <concepts::constructible_from_u8string_view T, typename ValT>
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(std::is_nothrow_constructible_v<T, std::u8string_view>) {
+  std::u8string_view str;
+  SIMDJSON_TRY(val.get_u8string().get(str));
+  if constexpr (requires { out.assign(str.data(), str.size()); }) {
+    // Copy straight into out (e.g., std::u8string), as for std::string above.
+    out.assign(str.data(), str.size());
+  } else {
+    out = T{str};
+  }
   return SUCCESS;
 }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T


+namespace details {
+// Whether to deserialize the elements of the container T directly into a new
+// element (emplace_one(out) and then get) rather than into a temporary that is
+// then moved into the container. A temporary of a trivially copyable type lives
+// in registers and is cheap to move, and value-initializing a small element in
+// place costs more than it saves (GCC zeroes it with rep stos). But moving a
+// large element, or a string (whose deserialization then needs a temporary
+// string and a move assignment of its own), is expensive.
+template <typename T>
+concept deserialize_in_place =
+    concepts::returns_reference<T> && requires(T &c) { c.pop_back(); } &&
+    !std::is_trivially_copyable_v<typename T::value_type> &&
+    (sizeof(typename T::value_type) > 32 || concepts::constructible_from_string_view<typename T::value_type>);
+
+// Removes the last element of the container on destruction while armed.
+template <typename T>
+struct pop_back_guard {
+  T &container;
+  bool armed{true};
+  ~pop_back_guard() {
+    if (armed) { container.pop_back(); }
+  }
+};
+} // namespace details
+
 /**
  * STL containers have several constructors including one that takes a single
  * size argument. Thus, some compilers (Visual Studio) will not be able to
@@ -68083,22 +84514,65 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
     SIMDJSON_TRY(val.get_array().get(arr));
   }

-  for (auto v : arr) {
-    if constexpr (concepts::returns_reference<T>) {
-      if (auto const err = v.get<value_type>().get(concepts::emplace_one(out));
-          err) {
-        // If an error occurs, the empty element that we just inserted gets
-        // removed. We're not using a temp variable because if T is a heavy
-        // type, we want the valid path to be the fast path and the slow path be
-        // the path that has errors in it.
-        if constexpr (requires { out.pop_back(); }) {
-          static_cast<void>(out.pop_back());
+  if constexpr (std::is_same_v<T, std::vector<value_type>> && !std::is_same_v<value_type, bool>) {
+    // Collect the elements in a per-thread scratch vector that keeps its
+    // capacity from call to call, then move them into out after reserving the
+    // exact size: out is allocated once instead of being regrown. A nested
+    // array of the same type finds the scratch busy and takes the paths below.
+    // Prior related work: jsonifier keeps a thread-local vector and sizes the
+    // caller's vector from that element count (parse_impl.hpp,
+    // https://github.com/nihilai-collective/Jsonifier).
+    struct scratch_space {
+      std::vector<value_type> elements{};
+      bool busy{false};
+    };
+    static thread_local scratch_space scratch;
+    if (!scratch.busy && out.empty()) {
+      struct release_scratch {
+        scratch_space &s;
+        T &out;
+        size_t parsed{0};
+        bool complete{false};
+        // On an error or an exception, out gets the elements parsed so far (as
+        // with the loops below), without allocating. Kept out of the hot path.
+        simdjson_never_inline void keep_parsed() noexcept {
+          s.elements.resize(parsed);
+          out.swap(s.elements);
         }
-        return err;
-      }
-    } else {
+        ~release_scratch() {
+          if (simdjson_unlikely(!complete)) { keep_parsed(); }
+          s.elements.clear();
+          // Do not hold on to the memory of a very large array.
+          if (s.elements.capacity() * sizeof(value_type) > (1 << 20)) { std::vector<value_type>().swap(s.elements); }
+          s.busy = false;
+        }
+      } release{scratch, out};
+      scratch.busy = true;
+      for (auto v : arr) {
+        SIMDJSON_TRY(v.get<value_type>(scratch.elements.emplace_back()));
+        release.parsed++;
+      }
+      out.reserve(release.parsed);
+      release.complete = true;
+      for (auto &e : scratch.elements) { out.emplace_back(std::move(e)); }
+      return SUCCESS;
+    }
+  }
+  if constexpr (details::deserialize_in_place<T>) {
+    for (auto v : arr) {
+      auto &slot = concepts::emplace_one(out);
+      // An error or an exception (a user tag_invoke may throw) must not leave
+      // a partially deserialized element behind.
+      details::pop_back_guard<T> guard{out};
+      SIMDJSON_TRY(v.get<value_type>(slot));
+      guard.armed = false;
+    }
+  } else {
+    for (auto v : arr) {
+      // Deserialize into a temporary first: an error or an exception (a user
+      // tag_invoke may throw) must not leave a default-constructed element behind.
       value_type temp;
-      if (auto const err = v.get<value_type>().get(temp); err) {
+      if (auto const err = v.get<value_type>(temp); err) {
         return err;
       }
       concepts::emplace_one(out, std::move(temp));
@@ -68139,7 +84613,7 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
 }

 template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, arm64::ondemand::object &obj, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, arm64::ondemand::object &obj, T &out) noexcept(false) {
   using value_type = typename std::remove_cvref_t<T>::mapped_type;

   out.clear();
@@ -68158,21 +84632,21 @@ error_code tag_invoke(deserialize_tag, arm64::ondemand::object &obj, T &out) noe
 }

 template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, arm64::ondemand::value &val, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, arm64::ondemand::value &val, T &out) noexcept(false) {
   arm64::ondemand::object obj;
   SIMDJSON_TRY(val.get_object().get(obj));
   return simdjson::deserialize(obj, out);
 }

 template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, arm64::ondemand::document &doc, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, arm64::ondemand::document &doc, T &out) noexcept(false) {
   arm64::ondemand::object obj;
   SIMDJSON_TRY(doc.get_object().get(obj));
   return simdjson::deserialize(obj, out);
 }

 template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, arm64::ondemand::document_reference &doc, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, arm64::ondemand::document_reference &doc, T &out) noexcept(false) {
   arm64::ondemand::object obj;
   SIMDJSON_TRY(doc.get_object().get(obj));
   return simdjson::deserialize(obj, out);
@@ -68183,10 +84657,6 @@ error_code tag_invoke(deserialize_tag, arm64::ondemand::document_reference &doc,
  * This CPO (Customization Point Object) will help deserialize into
  * smart pointers.
  *
- * If constructing T is nothrow, this conversion should be nothrow as well since
- * we return MEMALLOC if we're not able to allocate memory instead of throwing
- * the error message.
- *
  * @tparam T The type inside the smart pointer
  * @tparam ValT document/value type
  * @param val document/value
@@ -68194,7 +84664,7 @@ error_code tag_invoke(deserialize_tag, arm64::ondemand::document_reference &doc,
  * @return status of the conversion
  */
 template <concepts::smart_pointer T, typename ValT>
-error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(nothrow_deserializable<typename std::remove_cvref_t<T>::element_type, ValT>) {
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
   using element_type = typename std::remove_cvref_t<T>::element_type;

   // For better error messages, don't use these as constraints on
@@ -68206,12 +84676,13 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(nothrow_deser
       std::is_default_constructible_v<element_type>,
       "The specified type inside the unique_ptr must default constructible.");

-  auto ptr = new (std::nothrow) element_type();
-  if (ptr == nullptr) {
+  // Own the allocation before get(): a user tag_invoke may throw.
+  std::unique_ptr<element_type> ptr(new (std::nothrow) element_type());
+  if (!ptr) {
     return MEMALLOC;
   }
   SIMDJSON_TRY(val.template get<element_type>(*ptr));
-  out.reset(ptr);
+  out = std::move(ptr);
   return SUCCESS;
 }

@@ -68243,53 +84714,595 @@ error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept(nothrow_deser

 template <typename T>
 constexpr bool user_defined_type = (std::is_class_v<T>
-&& !std::is_same_v<T, std::string> && !std::is_same_v<T, std::string_view> && !concepts::optional_type<T> &&
-!concepts::appendable_containers<T>);
+&& !std::is_same_v<T, std::string> && !std::is_same_v<T, std::string_view>
+#if SIMDJSON_SUPPORTS_CHAR8_T
+// The char8_t string types are class types with no reflectable members, so
+// without this they would be taken for user structs and the reflection
+// overload below would compete with the u8 string overload in
+// std_deserialize.h, making every tag_invoke call on them ambiguous.
+&& !std::is_same_v<T, std::u8string> && !std::is_same_v<T, std::u8string_view>
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+&& !concepts::optional_type<T> &&
+!concepts::appendable_containers<T>
+// simdjson's own types (array, object, value, raw_json_string, number,
+// document, document_reference) have dedicated get<T>() specializations and
+// must never go through reflection.
+&& !is_builtin_deserializable_v<T>
+&& !std::is_same_v<T, arm64::ondemand::number>
+&& !std::is_same_v<T, arm64::ondemand::document>
+&& !std::is_same_v<T, arm64::ondemand::document_reference>);
+
+
+// key_selector_reflection_detail is defined unconditionally (it only requires
+// static reflection). It provides the compile-time machinery for building a
+// key_selector from a struct's members, the per-member helpers shared by every
+// deserialization path (they implement the annotations of annotations.h), the
+// ordered per-member path used by the opt-out build (see
+// deserialize_struct_ordered below), and the scan used for structs annotated
+// with deny_unknown_fields and as an automatic fallback (see
+// deserialize_struct_scan and keys_fit_selector below).
+namespace key_selector_reflection_detail {
+
+// A member participates if it is public, non-const, and not annotated with skip
+// or skip_deserializing.
+consteval bool is_eligible_member(std::meta::info mem) {
+  return !std::meta::is_const(mem) && std::meta::is_public(mem)
+      && !simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_tag)
+      && !simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_deserializing_tag);
+}
+
+// A field is a data member reached from T through a chain of members: usually a
+// direct member of T (a path of length one), or a member of a member annotated
+// with flatten (the members of a flattened structure are fields of the enclosing
+// one). A field is identified by the type member_path<m1, m2, ..., leaf>.
+template <std::meta::info First, std::meta::info... Rest>
+struct member_path {
+  // The data member holding the value; its annotations drive (de)serialization.
+  static constexpr std::meta::info leaf = [] {
+    std::meta::info members[] = {First, Rest...};
+    return members[sizeof...(Rest)];
+  }();
+  template <typename T>
+  static simdjson_inline constexpr auto &get(T &obj) noexcept {
+    if constexpr (sizeof...(Rest) == 0) {
+      return obj.[:First:];
+    } else {
+      return member_path<Rest...>::get(obj.[:First:]);
+    }
+  }
+};
+
+// True when T can be flattened: a structure deserialized member by member (not
+// a string, a container, an optional, a smart pointer, ...).
+template <typename T>
+constexpr bool flattenable_type = user_defined_type<T> && !concepts::string_view_keyed_map<T>
+    && !concepts::container_but_not_string<T> && !concepts::smart_pointer<T>;
+
+consteval void append_eligible_fields(std::meta::info type, std::vector<std::meta::info> &prefix,
+                                      std::vector<std::meta::info> &fields) {
+  for (std::meta::info mem : std::meta::nonstatic_data_members_of(type, std::meta::access_context::unchecked())) {
+    if (!is_eligible_member(mem)) { continue; }
+    prefix.push_back(std::meta::reflect_constant(mem));
+    if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::flatten_tag)) {
+      std::meta::info flattened = simdjson::detail::flattened_type(mem);
+      if (!std::meta::extract<bool>(std::meta::substitute(^^flattenable_type, {flattened}))) {
+        throw std::meta::exception(u8"simdjson::flatten requires a member whose type is a structure deserialized member by member", mem);
+      }
+      append_eligible_fields(flattened, prefix, fields);
+    } else {
+      fields.push_back(std::meta::substitute(^^member_path, prefix));
+    }
+    prefix.pop_back();
+  }
+}
+
+// The fields of `type` that participate in deserialization, in declaration order
+// (the fields of a flattened member take its place). A field's position in this
+// list is its "field index" below.
+consteval std::vector<std::meta::info> eligible_fields(std::meta::info type) {
+  std::vector<std::meta::info> prefix;
+  std::vector<std::meta::info> fields;
+  append_eligible_fields(type, prefix, fields);
+  return fields;
+}
+
+// The data member at the end of a field path.
+consteval std::meta::info field_leaf(std::meta::info path) {
+  return std::meta::extract<std::meta::info>(std::meta::template_arguments_of(path).back());
+}
+
+// Number of fields that participate in deserialization. A class can have zero
+// eligible fields (e.g. std::chrono::time_point, whose only data member is
+// private): an empty key_selector cannot be built, so the tag_invoke below
+// special-cases this count.
+template <typename T>
+consteval std::size_t eligible_field_count() {
+  return eligible_fields(^^T).size();
+}
+
+// The keys accepted for a member (or an enumerator): its JSON key followed by its
+// aliases. They are static strings, so they can be used as template arguments.
+consteval std::vector<const char *> accepted_keys_of(std::meta::info entity) {
+  std::vector<const char *> keys;
+  for (std::string_view key : simdjson::detail::json_key_names(entity)) {
+    bool repeated = false;
+    for (const char *previous : keys) { repeated = repeated || std::string_view(previous) == key; }
+    if (!repeated) { keys.push_back(std::define_static_string(key)); }
+  }
+  return keys;
+}
+
+// Every key accepted for T: the eligible fields in order, each followed by its
+// aliases. This is the key list of T's key_selector.
+consteval std::vector<const char *> accepted_keys(std::meta::info type) {
+  std::vector<const char *> keys;
+  for (std::meta::info path : eligible_fields(type)) {
+    for (const char *key : accepted_keys_of(field_leaf(path))) { keys.push_back(key); }
+  }
+  return keys;
+}
+
+// The field index of each entry of accepted_keys(type).
+consteval std::vector<std::size_t> accepted_key_fields(std::meta::info type) {
+  std::vector<std::size_t> key_fields;
+  std::vector<std::meta::info> fields = eligible_fields(type);
+  for (std::size_t i = 0; i < fields.size(); ++i) {
+    for (std::size_t j = 0; j < accepted_keys_of(field_leaf(fields[i])).size(); ++j) { key_fields.push_back(i); }
+  }
+  return key_fields;
+}
+
+// True when no two eligible fields of T accept the same key. Otherwise one JSON
+// key would have to fill several members: this is reported at compile time.
+template <typename T>
+consteval bool accepted_keys_are_distinct() {
+  std::vector<const char *> keys = accepted_keys(^^T);
+  for (std::size_t i = 0; i < keys.size(); ++i) {
+    for (std::size_t j = i + 1; j < keys.size(); ++j) {
+      if (std::string_view(keys[i]) == std::string_view(keys[j])) { return false; }
+    }
+  }
+  return true;
+}
+
+// True when some accepted key of T is written with escape sequences in JSON (a
+// double quote, a backslash or a control character). Such a key can only be
+// matched by comparing unescaped keys (deserialize_struct_scan): obj[key] and
+// the key_selector compare the raw bytes.
+template <typename T>
+consteval bool keys_need_unescaping() {
+  for (std::string_view key : accepted_keys(^^T)) {
+    for (char c : key) {
+      if (c == '\\' || c == '"' || static_cast<unsigned char>(c) < 0x20) { return true; }
+    }
+  }
+  return false;
+}

+// True when some eligible field of T has aliases: several selector keys may then
+// map to the same field.
+template <typename T>
+consteval bool has_aliases() {
+  return accepted_keys(^^T).size() != eligible_fields(^^T).size();
+}
+
+// True when `mem` is annotated with default_value or default_from (directly, or
+// through a default_value annotation on the enclosing structure).
+template <auto mem>
+consteval bool has_default() {
+  return simdjson::detail::has_annotation(mem, ^^simdjson::detail::default_value_tag)
+      || simdjson::detail::has_annotation(std::meta::parent_of(mem), ^^simdjson::detail::default_value_tag)
+      || simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::default_from_t) != std::meta::info{};
+}
+
+// True when a missing key for `mem` is not an error: optional members and
+// members with a default.
+template <auto mem>
+consteval bool may_be_absent() {
+  return concepts::optional_type<typename [: std::meta::type_of(mem) :]> || has_default<mem>();
+}
+
+// True when every eligible field of T is required (none may be absent). In that
+// case presence can be checked with a single match count instead of a per-field
+// "seen" array.
+template <typename T>
+consteval bool all_eligible_fields_required() {
+  bool all_required = true;
+  template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+    if constexpr (may_be_absent<[: path :]::leaf>()) {
+      all_required = false;
+    }
+  }
+  return all_required;
+}
+
+// `key` as a constevalutil::fixed_string usable as an NTTP.
+template <const char *key>
+consteval auto key_fixed_string() {
+  constexpr std::string_view key_view{ key };
+  char buffer[key_view.size() + 1] = {};
+  for (std::size_t i = 0; i < key_view.size(); ++i) { buffer[i] = key_view[i]; }
+  return constevalutil::fixed_string<key_view.size() + 1>(buffer);
+}
+
+// key_selector template arguments (one fixed_string per accepted key), in the
+// order of accepted_keys.
+template <typename T>
+consteval std::vector<std::meta::info> selector_key_args() {
+  std::vector<std::meta::info> args;
+  template for (constexpr const char *key : std::define_static_array(accepted_keys(^^T))) {
+    args.push_back(std::meta::reflect_constant(key_fixed_string<key>()));
+  }
+  return args;
+}
+
+// key_selector whose keys are exactly T's accepted keys (index i <-> i-th key).
+template <typename T>
+using selector_for = typename [: std::meta::substitute(
+    ^^arm64::ondemand::key_selector, selector_key_args<T>()) :];
+
+// True when T's accepted keys satisfy every key_selector requirement, so a
+// key_selector can be built for T without a compile-time error. This mirrors the
+// key_selector limits (see key_selector.h): at most 255 keys, each key non-empty
+// and at most 63 characters, no backslash / double-quote / null byte, and all
+// keys distinct. Keys with any other control character are excluded too: they
+// are escaped in JSON, so their raw bytes never match. When this returns false
+// the deserializer falls back to deserialize_struct_scan instead of failing to
+// compile.
+template <typename T>
+consteval bool keys_fit_selector() {
+  std::vector<std::string_view> keys;
+  for (const char *key : accepted_keys(^^T)) { keys.push_back(key); }
+  if (keys.size() > 255) { return false; }
+  for (std::size_t i = 0; i < keys.size(); ++i) {
+    if (keys[i].empty() || keys[i].size() > 63) { return false; }
+    for (char c : keys[i]) {
+      if (c == '\\' || c == '"' || static_cast<unsigned char>(c) < 0x20) { return false; }
+    }
+    for (std::size_t j = i + 1; j < keys.size(); ++j) {
+      if (keys[i] == keys[j]) { return false; }
+    }
+  }
+  return true;
+}
+
+// True when the class `adapter` declares a member named deserialize.
+consteval bool declares_deserialize(std::meta::info adapter) {
+  for (std::meta::info m : std::meta::members_of(adapter, std::meta::access_context::unchecked())) {
+    if (std::meta::has_identifier(m) && std::meta::identifier_of(m) == "deserialize") { return true; }
+  }
+  return false;
+}
+
+// Deserialize a JSON value into `target`, the storage of member `mem`, through
+// the member's with<Adapter> annotation when the adapter provides a deserialize
+// function.
+template <auto mem, typename ValueT, typename M>
+simdjson_warn_unused simdjson_inline error_code deserialize_member_value(ValueT &field_value, M &target) noexcept(false) {
+  constexpr std::meta::info with_type = simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::with_t);
+  if constexpr (with_type != std::meta::info{}) {
+    using adapter = typename [: with_type :]::adapter;
+    using ondemand_value = arm64::ondemand::value;
+    if constexpr (requires { { adapter::deserialize(field_value, target) } -> std::convertible_to<error_code>; }) {
+      return adapter::deserialize(field_value, target);
+    } else if constexpr (requires(ondemand_value &v) { { adapter::deserialize(v, target) } -> std::convertible_to<error_code>; }
+                         && requires { field_value.get_value(); }) {
+      // A transparent structure read from a document: the adapter takes an
+      // ondemand::value. A scalar document cannot be viewed as a value, so it
+      // reports SCALAR_DOCUMENT_AS_VALUE (an adapter taking auto& receives the
+      // document itself and has no such limitation).
+      ondemand_value v;
+      SIMDJSON_TRY(field_value.get_value().get(v));
+      return adapter::deserialize(v, target);
+    } else {
+      static_assert(!declares_deserialize(^^adapter),
+                    "the deserialize function of a simdjson::with adapter must be callable as "
+                    "Adapter::deserialize(simdjson::ondemand::value &, T &) and return an error_code");
+      return field_value.get(target);
+    }
+  } else {
+    return field_value.get(target);
+  }
+}
+
+// Deserialize the storage `target` of member `mem` from a JSON value.
+template <auto mem, typename ValueT, typename M>
+simdjson_warn_unused simdjson_inline error_code deserialize_member(ValueT &field_value, M &target) noexcept(false) {
+  if constexpr (has_default<mem>() && std::is_default_constructible_v<M> && std::is_move_assignable_v<M>) {
+    // A present key replaces the default value: deserialize into a fresh
+    // temporary so that, e.g., a container does not append to its default
+    // content, and a failure leaves the default untouched.
+    M value{};
+    SIMDJSON_TRY(deserialize_member_value<mem>(field_value, value));
+    target = std::move(value);
+    return SUCCESS;
+  } else {
+    return deserialize_member_value<mem>(field_value, target);
+  }
+}

+// Deserialize the field with the given field index.
+template <typename T>
+simdjson_warn_unused simdjson_inline error_code deserialize_field_at(
+    std::size_t field_index, arm64::ondemand::value field_value, T &out) noexcept(false) {
+  std::size_t counter = 0;
+  template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+    using field = [: path :];
+    if (field_index == counter) { return deserialize_member<field::leaf>(field_value, field::get(out)); }
+    ++counter;
+  }
+  return SUCCESS;
+}
+
+// Called when the key(s) of member `mem`, stored in `target`, are missing from
+// the JSON object: an error for a required member, a call to the default_from
+// factory, or nothing at all.
+template <auto mem, typename M>
+simdjson_warn_unused simdjson_inline error_code on_missing_member(M &target) noexcept(false) {
+  constexpr std::meta::info default_from_type = simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::default_from_t);
+  if constexpr (default_from_type != std::meta::info{}) {
+    target = [: default_from_type :]::factory();
+    return SUCCESS;
+  } else if constexpr (may_be_absent<mem>()) {
+    // For optional and default_value members, a missing key is not an error:
+    // leave the member at its current (default) value.
+    (void)target;
+    return SUCCESS;
+  } else {
+    (void)target;
+    return NO_SUCH_FIELD;
+  }
+}
+
+// Report or handle every field that was not seen in the JSON object.
+template <typename T, std::size_t N>
+simdjson_warn_unused simdjson_inline error_code handle_missing_fields(
+    const std::array<bool, N> &seen_field, T &out) noexcept(false) {
+  std::size_t counter = 0;
+  template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+    using field = [: path :];
+    if (!seen_field[counter]) { SIMDJSON_TRY(on_missing_member<field::leaf>(field::get(out))); }
+    ++counter;
+  }
+  return SUCCESS;
+}
+
+// Ordered, per-field deserialization: one obj[key] lookup per eligible field
+// (and per alias, until one is found). This is the opt-out path
+// (-DSIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION=1), except for structs with keys
+// that need unescaping (see keys_need_unescaping).
+template <typename T>
+simdjson_warn_unused error_code deserialize_struct_ordered(
+    arm64::ondemand::object &obj, T &out) noexcept(false) {
+  template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+    using field = [: path :];
+    arm64::ondemand::value field_value;
+    error_code error = NO_SUCH_FIELD;
+    template for (constexpr const char *key : std::define_static_array(accepted_keys_of(field::leaf))) {
+      if (error == NO_SUCH_FIELD) { error = obj[std::string_view(key)].get(field_value); }
+    }
+    if (error == NO_SUCH_FIELD) {
+      SIMDJSON_TRY(on_missing_member<field::leaf>(field::get(out)));
+    } else if (error) {
+      return error;
+    } else {
+      SIMDJSON_TRY(deserialize_member<field::leaf>(field_value, field::get(out)));
+    }
+  }
+  return SUCCESS;
+}
+
+// Appends the JSON keys that serialization writes for members of `type` that
+// deserialization cannot assign (const or non-public members, directly or
+// through flatten). `all` is true inside a flattened member that is itself
+// unassignable. Keys of skip_deserializing members are not included: they are
+// unknown keys, as documented.
+consteval void append_unassignable_keys(std::meta::info type, bool all, std::vector<const char *> &keys) {
+  for (std::meta::info mem : std::meta::nonstatic_data_members_of(type, std::meta::access_context::unchecked())) {
+    if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_tag)
+        || simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_serializing_tag)
+        || simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_deserializing_tag)) {
+      continue;
+    }
+    bool unassignable = all || !is_eligible_member(mem);
+    if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::flatten_tag)) {
+      append_unassignable_keys(simdjson::detail::flattened_type(mem), unassignable, keys);
+    } else if (unassignable) {
+      keys.push_back(std::define_static_string(simdjson::detail::json_key_name(mem)));
+    }
+  }
+}
+
+consteval std::vector<const char *> unassignable_keys(std::meta::info type) {
+  std::vector<const char *> keys;
+  append_unassignable_keys(type, false, keys);
+  return keys;
+}
+
+// Deserialization by a single pass over every field of the object, comparing
+// unescaped keys. It is used for structs annotated with deny_unknown_fields
+// (DenyUnknown = true), where a key that does not map to an eligible field is
+// reported as UNKNOWN_FIELD, and as the fallback for structs whose keys the
+// key_selector or obj[key] cannot match (see keys_fit_selector and
+// keys_need_unescaping). As with object::for_each, the first occurrence of a
+// field wins (later duplicates, or aliases of a field already seen, are ignored).
+template <bool DenyUnknown, typename T>
+simdjson_warn_unused error_code deserialize_struct_scan(
+    arm64::ondemand::object &obj, T &out) noexcept(false) {
+  static constexpr auto keys = std::define_static_array(accepted_keys(^^T));
+  static constexpr auto key_fields = std::define_static_array(accepted_key_fields(^^T));
+  std::array<bool, eligible_field_count<T>()> seen_field{};
+  for (auto field_result : obj) {
+    arm64::ondemand::field json_field;
+    SIMDJSON_TRY(std::move(field_result).get(json_field));
+    std::string_view key;
+    SIMDJSON_TRY(json_field.unescaped_key().get(key));
+    std::size_t key_index = keys.size();
+    for (std::size_t i = 0; i < keys.size(); ++i) {
+      if (key == std::string_view(keys[i])) { key_index = i; break; }
+    }
+    if (key_index == keys.size()) {
+      if constexpr (DenyUnknown) {
+        // A key that T itself serializes (e.g. of a const member) is not
+        // unknown: a serialized value must parse back.
+        static constexpr auto ignored_keys = std::define_static_array(unassignable_keys(^^T));
+        bool ignored = false;
+        for (const char *ignored_key : ignored_keys) {
+          if (key == std::string_view(ignored_key)) { ignored = true; break; }
+        }
+        if (!ignored) { return UNKNOWN_FIELD; }
+      }
+      continue;
+    }
+    const std::size_t field_index = key_fields[key_index];
+    if (seen_field[field_index]) { continue; }
+    seen_field[field_index] = true;
+    SIMDJSON_TRY(deserialize_field_at(field_index, json_field.value(), out));
+  }
+  return handle_missing_fields(seen_field, out);
+}
+
+template <typename T>
+consteval bool is_transparent() {
+  return simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag);
+}
+
+} // namespace key_selector_reflection_detail
+
+// Deserialize a reflected struct. By default this builds a compile-time
+// key_selector from the struct's members and walks each object once with
+// object::for_each (perfect-hash key matching), instead of one obj[key] lookup
+// per member. Other paths are used instead:
+//   - globally, the ordered per-member path when defining
+//     -DSIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION=1;
+//   - automatically and per-type, a scan of the object comparing unescaped keys
+//     when the struct's keys do not fit the key_selector limits (see
+//     keys_fit_selector) or cannot be compared raw (see keys_need_unescaping),
+//     so that long member names and the like keep compiling rather than
+//     tripping a static_assert.
+// Structs annotated with deny_unknown_fields always use the strict scan, and
+// structs annotated with transparent are deserialized as their single member.
+//
+// noexcept(false): a member's tag_invoke may throw and the exception must reach
+// the caller of get<T>().
 template <typename T, typename ValT>
   requires(user_defined_type<T> && std::is_class_v<T>)
-error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
+  if constexpr (key_selector_reflection_detail::is_transparent<T>()) {
+    constexpr auto mem = simdjson::detail::transparent_member(^^T);
+    if constexpr (std::is_same_v<std::remove_cvref_t<ValT>, arm64::ondemand::object>) {
+      // We were handed an object: only a structure can be deserialized from it.
+      if constexpr (user_defined_type<typename [: std::meta::type_of(mem) :]>) {
+        return tag_invoke(deserialize_tag{}, val, out.[:mem:]);
+      } else {
+        return INCORRECT_TYPE;
+      }
+    } else {
+      return key_selector_reflection_detail::deserialize_member<mem>(val, out.[:mem:]);
+    }
+  } else {
+  static_assert(key_selector_reflection_detail::accepted_keys_are_distinct<T>(),
+                "two members of this structure accept the same JSON key (check rename, alias, "
+                "rename_all and flatten)");
   arm64::ondemand::object obj;
   if constexpr (std::is_same_v<std::remove_cvref_t<ValT>, arm64::ondemand::object>) {
     obj = val;
   } else {
     SIMDJSON_TRY(val.get_object().get(obj));
   }
-  template for (constexpr auto mem : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
-    if constexpr (!std::meta::is_const(mem) && std::meta::is_public(mem)) {
-      constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
-      if constexpr (concepts::optional_type<decltype(out.[:mem:])>) {
-        // for optional members, it's ok if the key is missing
-        auto error = obj[key].get(out.[:mem:]);
-        if (error && error != NO_SUCH_FIELD) {
-          if(error == NO_SUCH_FIELD) {
-            out.[:mem:].reset();
-            continue;
-          }
-          return error;
-        }
-      } else {
-        // for non-optional members, the key must be present
-        SIMDJSON_TRY(obj[key].get(out.[:mem:]));
+  if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::deny_unknown_fields_tag)) {
+    return key_selector_reflection_detail::deserialize_struct_scan<true>(obj, out);
+  } else {
+#if defined(SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION) && SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION
+  // Opt-out build: use the ordered per-member path, unless obj[key] cannot
+  // match T's keys.
+  if constexpr (key_selector_reflection_detail::keys_need_unescaping<T>()) {
+    return key_selector_reflection_detail::deserialize_struct_scan<false>(obj, out);
+  } else {
+    return key_selector_reflection_detail::deserialize_struct_ordered(obj, out);
+  }
+#else
+  if constexpr (key_selector_reflection_detail::eligible_field_count<T>() == 0) {
+    // No fields to deserialize: an empty key_selector cannot be built, so just
+    // validate that the input is an object (done above) and succeed. Mirrors the
+    // ordered per-member path, which iterates over zero members.
+    (void)out;
+    (void)obj;
+    return SUCCESS;
+  } else if constexpr (!key_selector_reflection_detail::keys_fit_selector<T>()) {
+    // Automatic fallback: T's accepted keys do not fit the key_selector limits
+    // (e.g. a member name longer than 63 characters, or a key with a double
+    // quote), so building a selector would be a compile error. Scan the object
+    // instead, so the default never breaks a struct that the opt-out path would
+    // accept.
+    return key_selector_reflection_detail::deserialize_struct_scan<false>(obj, out);
+  } else {
+  using selector = key_selector_reflection_detail::selector_for<T>;
+  if constexpr (key_selector_reflection_detail::all_eligible_fields_required<T>()
+                && !key_selector_reflection_detail::has_aliases<T>()) {
+    // Fast path: every member is required and has a single key. A single
+    // for_each pass parses each matched field; the returned match count then
+    // tells us whether every member was present (matched_count ==
+    // selector::size()) without a per-member "seen" array. A value-parse error
+    // (e.g. a type mismatch) is propagated by for_each.
+    auto walk = obj.template for_each<selector>(
+        [&](std::size_t matched_index, arm64::ondemand::value field_value) -> error_code {
+      std::size_t counter = 0;
+      template for (constexpr auto path : std::define_static_array(key_selector_reflection_detail::eligible_fields(^^T))) {
+        using field = [: path :];
+        if (matched_index == counter) { return key_selector_reflection_detail::deserialize_member<field::leaf>(field_value, field::get(out)); }
+        ++counter;
       }
-    }
-  };
-  return simdjson::SUCCESS;
-}
-
-// Support for enum deserialization - deserialize from string representation using expand approach from P2996R12
+      return SUCCESS;
+    });
+    if (walk.error) { return walk.error; }
+    // A missing required member shows up as a short match count and is reported as
+    // NO_SUCH_FIELD, mirroring the ordered obj[key] path.
+    if (walk.matched_count != selector::size()) { return NO_SUCH_FIELD; }
+    return SUCCESS;
+  } else {
+    static constexpr auto key_fields = std::define_static_array(key_selector_reflection_detail::accepted_key_fields(^^T));
+    std::array<bool, key_selector_reflection_detail::eligible_field_count<T>()> seen_field{};
+    // Single pass over the object: each field whose key matches a member (or one
+    // of its aliases) yields its selector index, which we map back to the
+    // corresponding member. The first key seen for a member wins. The callback
+    // returns an error_code so that a value-parse error (e.g. a type mismatch on
+    // a matched field) is propagated by for_each instead of being silently dropped.
+    error_code walk_error = obj.template for_each<selector>(
+        [&](std::size_t matched_index, arm64::ondemand::value field_value) -> error_code {
+      const std::size_t field_index = key_fields[matched_index];
+      if (seen_field[field_index]) { return SUCCESS; }
+      seen_field[field_index] = true;
+      return key_selector_reflection_detail::deserialize_field_at(field_index, field_value, out);
+    });
+    if (walk_error) { return walk_error; }
+    // Required members must be present: a missing one is reported as
+    // NO_SUCH_FIELD, mirroring the ordered obj[key] path. Optional and defaulted
+    // members may be absent.
+    return key_selector_reflection_detail::handle_missing_fields(seen_field, out);
+  }
+  }
+#endif // SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION
+  }
+  }
+}
+
+// Support for enum deserialization - deserialize from string representation using
+// expand approach from P2996R12. The accepted strings are the enumerator's JSON key
+// (see rename and rename_all) and its aliases.
 template <typename T, typename ValT>
   requires(std::is_enum_v<T>)
 error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
 #if SIMDJSON_STATIC_REFLECTION
   std::string_view str;
   SIMDJSON_TRY(val.get_string().get(str));
-  constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
+  static constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
   template for (constexpr auto enum_val : enumerators) {
-    if (str == std::meta::identifier_of(enum_val)) {
-      out = [:enum_val:];
-      return SUCCESS;
+    template for (constexpr const char *key : std::define_static_array(key_selector_reflection_detail::accepted_keys_of(enum_val))) {
+      if (str == std::string_view(key)) {
+        out = [:enum_val:];
+        return SUCCESS;
+      }
     }
   };

@@ -68305,33 +85318,25 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {

 template <typename simdjson_value, typename T>
   requires(user_defined_type<std::remove_cvref_t<T>>)
-error_code tag_invoke(deserialize_tag, simdjson_value &val, std::unique_ptr<T> &out) noexcept {
-  if (!out) {
-    out = std::make_unique<T>();
-    if (!out) {
-      return MEMALLOC;
-    }
-  }
-  if (auto err = val.get(*out)) {
-    out.reset();
-    return err;
+error_code tag_invoke(deserialize_tag, simdjson_value &val, std::unique_ptr<T> &out) noexcept(false) {
+  std::unique_ptr<T> ptr(new (std::nothrow) T());
+  if (!ptr) {
+    return MEMALLOC;
   }
+  SIMDJSON_TRY(val.get(*ptr));
+  out = std::move(ptr);
   return SUCCESS;
 }

 template <typename simdjson_value, typename T>
   requires(user_defined_type<std::remove_cvref_t<T>>)
-error_code tag_invoke(deserialize_tag, simdjson_value &val, std::shared_ptr<T> &out) noexcept {
-  if (!out) {
-    out = std::make_shared<T>();
-    if (!out) {
-      return MEMALLOC;
-    }
-  }
-  if (auto err = val.get(*out)) {
-    out.reset();
-    return err;
+error_code tag_invoke(deserialize_tag, simdjson_value &val, std::shared_ptr<T> &out) noexcept(false) {
+  std::shared_ptr<T> ptr(new (std::nothrow) T());
+  if (!ptr) {
+    return MEMALLOC;
   }
+  SIMDJSON_TRY(val.get(*ptr));
+  out = std::move(ptr);
   return SUCCESS;
 }

@@ -68643,9 +85648,17 @@ simdjson_inline simdjson_result<array> array::started(value_iterator &iter) noex
   return array(iter);
 }

-simdjson_inline simdjson_result<array_iterator> array::begin() noexcept {
+simdjson_inline simdjson_result<array_iterator> array::begin() & noexcept {
 #if SIMDJSON_DEVELOPMENT_CHECKS
-  if (!iter.is_at_iterator_start()) { return OUT_OF_ORDER_ITERATION; }
+  if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
+  return array_iterator(iter, this);
+#endif
+  return array_iterator(iter);
+}
+simdjson_inline simdjson_result<array_iterator> array::begin() && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  // The array is a temporary that the iterator may outlive: do not lock it.
+  if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
 #endif
   return array_iterator(iter);
 }
@@ -68672,6 +85685,9 @@ simdjson_inline simdjson_result<std::string_view> array::raw_json() noexcept {
 SIMDJSON_PUSH_DISABLE_WARNINGS
 SIMDJSON_DISABLE_STRICT_OVERFLOW_WARNING
 simdjson_inline simdjson_result<size_t> array::count_elements() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   size_t count{0};
   // Important: we do not consume any of the values.
   for(simdjson_unused auto v : *this) { count++; }
@@ -68685,6 +85701,9 @@ simdjson_inline simdjson_result<size_t> array::count_elements() & noexcept {
 SIMDJSON_POP_DISABLE_WARNINGS

 simdjson_inline simdjson_result<bool> array::is_empty() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   bool is_not_empty;
   auto error = iter.reset_array().get(is_not_empty);
   if(error) { return error; }
@@ -68692,31 +85711,30 @@ simdjson_inline simdjson_result<bool> array::is_empty() & noexcept {
 }

 inline simdjson_result<bool> array::reset() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   return iter.reset_array();
 }

+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline void array::set_locked(bool _locked) noexcept {
+  locked = _locked;
+}
+#endif
+
 inline simdjson_result<value> array::at_pointer(std::string_view json_pointer) noexcept {
-  if (json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
+  // An empty pointer has no json_pointer[0]: with a default-constructed
+  // std::string_view, reading it dereferences a null pointer.
+  if (json_pointer.empty() || json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
   json_pointer = json_pointer.substr(1);
   // - means "the append position" or "the element after the end of the array"
   // We don't support this, because we're returning a real element, not a position.
   if (json_pointer == "-") { return INDEX_OUT_OF_BOUNDS; }

-  // Read the array index
   size_t array_index = 0;
   size_t i;
-  for (i = 0; i < json_pointer.length() && json_pointer[i] != '/'; i++) {
-    uint8_t digit = uint8_t(json_pointer[i] - '0');
-    // Check for non-digit in array index. If it's there, we're trying to get a field in an object
-    if (digit > 9) { return INCORRECT_TYPE; }
-    array_index = array_index*10 + digit;
-  }
-
-  // 0 followed by other digits is invalid
-  if (i > 1 && json_pointer[0] == '0') { return INVALID_JSON_POINTER; } // "JSON pointer array index has other characters after 0"
-
-  // Empty string is invalid; so is a "/" with no digits before it
-  if (i == 0) { return INVALID_JSON_POINTER; } // "Empty string in JSON pointer array index"
+  SIMDJSON_TRY(internal::parse_json_pointer_array_index(json_pointer, array_index, i));
   // Get the child
   auto child = at(array_index);
   // If there is an error, it ends here
@@ -68790,6 +85808,9 @@ inline error_code array::for_each_at_path_with_wildcard(std::string_view json_pa
 }

 simdjson_inline simdjson_result<value> array::at(size_t index) noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   size_t i = 0;
   for (auto value : *this) {
     if (i == index) { return value; }
@@ -68819,10 +85840,14 @@ simdjson_inline simdjson_result<arm64::ondemand::array>::simdjson_result(
 {
 }

-simdjson_inline simdjson_result<arm64::ondemand::array_iterator> simdjson_result<arm64::ondemand::array>::begin() noexcept {
+simdjson_inline simdjson_result<arm64::ondemand::array_iterator> simdjson_result<arm64::ondemand::array>::begin() & noexcept {
   if (error()) { return error(); }
   return first.begin();
 }
+simdjson_inline simdjson_result<arm64::ondemand::array_iterator> simdjson_result<arm64::ondemand::array>::begin() && noexcept {
+  if (error()) { return error(); }
+  return std::move(first).begin();
+}
 simdjson_inline simdjson_result<arm64::ondemand::array_iterator> simdjson_result<arm64::ondemand::array>::end() noexcept {
   if (error()) { return error(); }
   return first.end();
@@ -68885,6 +85910,59 @@ simdjson_inline array_iterator::array_iterator(const value_iterator &_iter) noex
   : iter{_iter}
 {}

+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline array_iterator::array_iterator(const value_iterator &_iter, array* _parent) noexcept
+  : parent{_parent}, iter{_iter}
+{
+  if (parent) parent->set_locked(true);
+}
+
+simdjson_inline array_iterator::~array_iterator() noexcept
+{
+  if (parent) parent->set_locked(false);
+}
+
+simdjson_inline array_iterator::array_iterator(array_iterator&& other) noexcept
+  : has_been_referenced{other.has_been_referenced},
+    parent{other.parent},
+    iter{std::move(other.iter)}
+{
+  other.parent = nullptr;
+}
+
+simdjson_inline array_iterator& array_iterator::operator=(array_iterator&& other) noexcept {
+  if (this != &other)
+  {
+    if (parent)
+      parent->set_locked(false);
+    has_been_referenced = other.has_been_referenced;
+    parent = other.parent;
+    iter = std::move(other.iter);
+
+    other.parent = nullptr;
+  }
+  return *this;
+}
+
+simdjson_inline array_iterator::array_iterator(const array_iterator& other) noexcept
+  : has_been_referenced{other.has_been_referenced},
+    parent{nullptr},
+    iter{other.iter}
+{}
+
+simdjson_inline array_iterator& array_iterator::operator=(const array_iterator& other) noexcept {
+  if (this != &other)
+  {
+    if (parent)
+      parent->set_locked(false);
+    has_been_referenced = other.has_been_referenced;
+    parent = nullptr;
+    iter = other.iter;
+  }
+  return *this;
+}
+#endif
+
 simdjson_inline simdjson_result<value> array_iterator::operator*() noexcept {
 #if SIMDJSON_DEVELOPMENT_CHECKS
    SIMDJSON_ASSUME(!has_been_referenced);
@@ -68980,6 +86058,41 @@ namespace simdjson {
 namespace arm64 {
 namespace ondemand {

+/**
+ * @private Narrows the result of get_uint64() to a smaller unsigned integer type T.
+ * @returns NUMBER_OUT_OF_RANGE if the value does not fit in a T.
+ */
+template <typename T>
+simdjson_inline simdjson_result<T> narrow_integer(simdjson_result<uint64_t> wide) noexcept {
+  uint64_t result;
+  SIMDJSON_TRY(std::move(wide).get(result));
+  if (result > static_cast<uint64_t>((std::numeric_limits<T>::max)())) { return NUMBER_OUT_OF_RANGE; }
+  return static_cast<T>(result);
+}
+
+/**
+ * @private Narrows the result of get_int64() to a smaller signed integer type T.
+ * @returns NUMBER_OUT_OF_RANGE if the value does not fit in a T.
+ */
+template <typename T>
+simdjson_inline simdjson_result<T> narrow_integer(simdjson_result<int64_t> wide) noexcept {
+  int64_t result;
+  SIMDJSON_TRY(std::move(wide).get(result));
+  if (result > static_cast<int64_t>((std::numeric_limits<T>::max)()) || result < static_cast<int64_t>((std::numeric_limits<T>::min)())) { return NUMBER_OUT_OF_RANGE; }
+  return static_cast<T>(result);
+}
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+// get_float32() parses with get_float(), so float must be binary32.
+static_assert(std::numeric_limits<float>::is_iec559 && std::numeric_limits<float>::digits == 24,
+              "get_float32() requires float to be IEEE binary32");
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+// get_float64() parses with get_double(), so double must be binary64.
+static_assert(std::numeric_limits<double>::is_iec559 && std::numeric_limits<double>::digits == 53,
+              "get_float64() requires double to be IEEE binary64");
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
+
 simdjson_inline value::value(const value_iterator &_iter) noexcept
   : iter{_iter}
 {
@@ -69011,6 +86124,13 @@ simdjson_inline simdjson_result<raw_json_string> value::get_raw_json_string() no
 simdjson_inline simdjson_result<std::string_view> value::get_string(bool allow_replacement) noexcept {
   return iter.get_string(allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> value::get_u8string(bool allow_replacement) noexcept {
+  std::string_view content;
+  SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+  return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code value::get_string(string_type& receiver, bool allow_replacement) noexcept {
   return iter.get_string(receiver, allow_replacement);
@@ -69024,6 +86144,12 @@ simdjson_inline simdjson_result<double> value::get_double() noexcept {
 simdjson_inline simdjson_result<double> value::get_double_in_string() noexcept {
   return iter.get_double_in_string();
 }
+simdjson_inline simdjson_result<float> value::get_float() noexcept {
+  return iter.get_float();
+}
+simdjson_inline simdjson_result<float> value::get_float_in_string() noexcept {
+  return iter.get_float_in_string();
+}
 simdjson_inline simdjson_result<uint64_t> value::get_uint64() noexcept {
   return iter.get_uint64();
 }
@@ -69037,17 +86163,37 @@ simdjson_inline simdjson_result<int64_t> value::get_int64_in_string() noexcept {
   return iter.get_int64_in_string();
 }
 simdjson_inline simdjson_result<uint32_t> value::get_uint32() noexcept {
-  uint64_t result;
-  SIMDJSON_TRY(get_uint64().get(result));
-  if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<uint32_t>(result);
+  return narrow_integer<uint32_t>(get_uint64());
 }
 simdjson_inline simdjson_result<int32_t> value::get_int32() noexcept {
-  int64_t result;
-  SIMDJSON_TRY(get_int64().get(result));
-  if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<int32_t>(result);
+  return narrow_integer<int32_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint16_t> value::get_uint16() noexcept {
+  return narrow_integer<uint16_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int16_t> value::get_int16() noexcept {
+  return narrow_integer<int16_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint8_t> value::get_uint8() noexcept {
+  return narrow_integer<uint8_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int8_t> value::get_int8() noexcept {
+  return narrow_integer<int8_t>(get_int64());
 }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> value::get_float32() noexcept {
+  float result;
+  SIMDJSON_TRY(get_float().get(result));
+  return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> value::get_float64() noexcept {
+  double result;
+  SIMDJSON_TRY(get_double().get(result));
+  return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<bool> value::get_bool() noexcept {
   return iter.get_bool();
 }
@@ -69059,12 +86205,26 @@ template<> simdjson_inline simdjson_result<array> value::get() noexcept { return
 template<> simdjson_inline simdjson_result<object> value::get() noexcept { return get_object(); }
 template<> simdjson_inline simdjson_result<raw_json_string> value::get() noexcept { return get_raw_json_string(); }
 template<> simdjson_inline simdjson_result<std::string_view> value::get() noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> value::get() noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_inline simdjson_result<number> value::get() noexcept { return get_number(); }
 template<> simdjson_inline simdjson_result<double> value::get() noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> value::get() noexcept { return get_float(); }
 template<> simdjson_inline simdjson_result<uint64_t> value::get() noexcept { return get_uint64(); }
 template<> simdjson_inline simdjson_result<int64_t> value::get() noexcept { return get_int64(); }
 template<> simdjson_inline simdjson_result<uint32_t> value::get() noexcept { return get_uint32(); }
 template<> simdjson_inline simdjson_result<int32_t> value::get() noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> value::get() noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> value::get() noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> value::get() noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> value::get() noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> value::get() noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> value::get() noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<> simdjson_inline simdjson_result<bool> value::get() noexcept { return get_bool(); }


@@ -69072,12 +86232,26 @@ template<> simdjson_warn_unused simdjson_inline error_code value::get(array& out
 template<> simdjson_warn_unused simdjson_inline error_code value::get(object& out) noexcept { return get_object().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(raw_json_string& out) noexcept { return get_raw_json_string().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(std::string_view& out) noexcept { return get_string(false).get(out); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::u8string_view& out) noexcept { return get_u8string(false).get(out); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_warn_unused simdjson_inline error_code value::get(number& out) noexcept { return get_number().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(double& out) noexcept { return get_double().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(float& out) noexcept { return get_float().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(uint64_t& out) noexcept { return get_uint64().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(int64_t& out) noexcept { return get_int64().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(uint32_t& out) noexcept { return get_uint32().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(int32_t& out) noexcept { return get_int32().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(uint16_t& out) noexcept { return get_uint16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(int16_t& out) noexcept { return get_int16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(uint8_t& out) noexcept { return get_uint8().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(int8_t& out) noexcept { return get_int8().get(out); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::float32_t& out) noexcept { return get_float32().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::float64_t& out) noexcept { return get_float64().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<>  simdjson_warn_unused simdjson_inline error_code value::get(bool& out) noexcept { return get_bool().get(out); }

 #if SIMDJSON_EXCEPTIONS
@@ -69246,6 +86420,9 @@ inline bool is_pointer_well_formed(std::string_view json_pointer) noexcept {
 }

 simdjson_inline simdjson_result<value> value::at_pointer(std::string_view json_pointer) noexcept {
+  // The empty JSON Pointer refers to the whole value (RFC 6901), as in
+  // document::at_pointer.
+  if (json_pointer.empty()) { return value(iter); }
   json_type t;
   SIMDJSON_TRY(type().get(t));
   switch (t)
@@ -69283,6 +86460,10 @@ template <typename Func>
 template <typename Func>
 #endif
 inline error_code value::for_each_at_path_with_wildcard(std::string_view json_path, Func&& callback) noexcept {
+  // Every recursive step of for_each_at_path_with_wildcard goes through this
+  // function, and each one descends one level into the document. A path with
+  // many segments applied to a deeply nested document would otherwise recurse
+  // without bound and overflow the stack.
   if (size_t(iter.depth()) >= iter.json_iter().parser->max_depth()) { return DEPTH_ERROR; }
   json_type t;
   SIMDJSON_TRY(type().get(t));
@@ -69396,10 +86577,46 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<arm64::ondemand::value>
   if (error()) { return error(); }
   return first.get_int32();
 }
+simdjson_inline simdjson_result<uint16_t> simdjson_result<arm64::ondemand::value>::get_uint16() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<arm64::ondemand::value>::get_int16() noexcept {
+  if (error()) { return error(); }
+  return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<arm64::ondemand::value>::get_uint8() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<arm64::ondemand::value>::get_int8() noexcept {
+  if (error()) { return error(); }
+  return first.get_int8();
+}
 simdjson_inline simdjson_result<double> simdjson_result<arm64::ondemand::value>::get_double() noexcept {
   if (error()) { return error(); }
   return first.get_double();
 }
+simdjson_inline simdjson_result<float> simdjson_result<arm64::ondemand::value>::get_float() noexcept {
+  if (error()) { return error(); }
+  return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<arm64::ondemand::value>::get_float_in_string() noexcept {
+  if (error()) { return error(); }
+  return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<arm64::ondemand::value>::get_float32() noexcept {
+  if (error()) { return error(); }
+  return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<arm64::ondemand::value>::get_float64() noexcept {
+  if (error()) { return error(); }
+  return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<double> simdjson_result<arm64::ondemand::value>::get_double_in_string() noexcept {
   if (error()) { return error(); }
   return first.get_double_in_string();
@@ -69408,6 +86625,12 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<arm64::ondeman
   if (error()) { return error(); }
   return first.get_string(allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<arm64::ondemand::value>::get_u8string(bool allow_replacement) noexcept {
+  if (error()) { return error(); }
+  return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_inline error_code simdjson_result<arm64::ondemand::value>::get_string(string_type& receiver, bool allow_replacement) noexcept {
   if (error()) { return error(); }
@@ -69436,11 +86659,23 @@ template<> simdjson_inline error_code simdjson_result<arm64::ondemand::value>::g
   return SUCCESS;
 }

-template<typename T> simdjson_inline simdjson_result<T> simdjson_result<arm64::ondemand::value>::get() noexcept {
+template<typename T> simdjson_inline simdjson_result<T> simdjson_result<arm64::ondemand::value>::get()
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, arm64::ondemand::value>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>();
 }
-template<typename T> simdjson_inline error_code simdjson_result<arm64::ondemand::value>::get(T &out) noexcept {
+template<typename T> simdjson_inline error_code simdjson_result<arm64::ondemand::value>::get(T &out)
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, arm64::ondemand::value>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>(out);
 }
@@ -69710,16 +86945,22 @@ simdjson_inline simdjson_result<int64_t> document::get_int64_in_string() noexcep
   return get_root_value_iterator().get_root_int64_in_string(true);
 }
 simdjson_inline simdjson_result<uint32_t> document::get_uint32() noexcept {
-  uint64_t result;
-  SIMDJSON_TRY(get_uint64().get(result));
-  if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<uint32_t>(result);
+  return narrow_integer<uint32_t>(get_uint64());
 }
 simdjson_inline simdjson_result<int32_t> document::get_int32() noexcept {
-  int64_t result;
-  SIMDJSON_TRY(get_int64().get(result));
-  if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<int32_t>(result);
+  return narrow_integer<int32_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint16_t> document::get_uint16() noexcept {
+  return narrow_integer<uint16_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int16_t> document::get_int16() noexcept {
+  return narrow_integer<int16_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint8_t> document::get_uint8() noexcept {
+  return narrow_integer<uint8_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int8_t> document::get_int8() noexcept {
+  return narrow_integer<int8_t>(get_int64());
 }
 simdjson_inline simdjson_result<double> document::get_double() noexcept {
   return get_root_value_iterator().get_root_double(true);
@@ -69727,9 +86968,36 @@ simdjson_inline simdjson_result<double> document::get_double() noexcept {
 simdjson_inline simdjson_result<double> document::get_double_in_string() noexcept {
   return get_root_value_iterator().get_root_double_in_string(true);
 }
+simdjson_inline simdjson_result<float> document::get_float() noexcept {
+  return get_root_value_iterator().get_root_float(true);
+}
+simdjson_inline simdjson_result<float> document::get_float_in_string() noexcept {
+  return get_root_value_iterator().get_root_float_in_string(true);
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> document::get_float32() noexcept {
+  float result;
+  SIMDJSON_TRY(get_float().get(result));
+  return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> document::get_float64() noexcept {
+  double result;
+  SIMDJSON_TRY(get_double().get(result));
+  return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<std::string_view> document::get_string(bool allow_replacement) noexcept {
   return get_root_value_iterator().get_root_string(true, allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> document::get_u8string(bool allow_replacement) noexcept {
+  std::string_view content;
+  SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+  return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code document::get_string(string_type& receiver, bool allow_replacement) noexcept {
   return get_root_value_iterator().get_root_string(receiver, true, allow_replacement);
@@ -69751,11 +87019,25 @@ template<> simdjson_inline simdjson_result<array> document::get() & noexcept { r
 template<> simdjson_inline simdjson_result<object> document::get() & noexcept { return get_object(); }
 template<> simdjson_inline simdjson_result<raw_json_string> document::get() & noexcept { return get_raw_json_string(); }
 template<> simdjson_inline simdjson_result<std::string_view> document::get() & noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> document::get() & noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_inline simdjson_result<double> document::get() & noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> document::get() & noexcept { return get_float(); }
 template<> simdjson_inline simdjson_result<uint64_t> document::get() & noexcept { return get_uint64(); }
 template<> simdjson_inline simdjson_result<int64_t> document::get() & noexcept { return get_int64(); }
 template<> simdjson_inline simdjson_result<uint32_t> document::get() & noexcept { return get_uint32(); }
 template<> simdjson_inline simdjson_result<int32_t> document::get() & noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> document::get() & noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> document::get() & noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> document::get() & noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> document::get() & noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> document::get() & noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> document::get() & noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<> simdjson_inline simdjson_result<bool> document::get() & noexcept { return get_bool(); }
 template<> simdjson_inline simdjson_result<value> document::get() & noexcept { return get_value(); }

@@ -69763,17 +87045,35 @@ template<> simdjson_warn_unused simdjson_inline error_code document::get(array&
 template<> simdjson_warn_unused simdjson_inline error_code document::get(object& out) & noexcept { return get_object().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(raw_json_string& out) & noexcept { return get_raw_json_string().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(std::string_view& out) & noexcept { return get_string(false).get(out); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::u8string_view& out) & noexcept { return get_u8string(false).get(out); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_warn_unused simdjson_inline error_code document::get(double& out) & noexcept { return get_double().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(float& out) & noexcept { return get_float().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(uint64_t& out) & noexcept { return get_uint64().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(int64_t& out) & noexcept { return get_int64().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(uint32_t& out) & noexcept { return get_uint32().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(int32_t& out) & noexcept { return get_int32().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(uint16_t& out) & noexcept { return get_uint16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(int16_t& out) & noexcept { return get_int16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(uint8_t& out) & noexcept { return get_uint8().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(int8_t& out) & noexcept { return get_int8().get(out); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::float32_t& out) & noexcept { return get_float32().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::float64_t& out) & noexcept { return get_float64().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<> simdjson_warn_unused simdjson_inline error_code document::get(bool& out) & noexcept { return get_bool().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(value& out) & noexcept { return get_value().get(out); }

 template<> simdjson_deprecated simdjson_inline simdjson_result<raw_json_string> document::get() && noexcept { return get_raw_json_string(); }
 template<> simdjson_deprecated simdjson_inline simdjson_result<std::string_view> document::get() && noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_deprecated simdjson_inline simdjson_result<std::u8string_view> document::get() && noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_deprecated simdjson_inline simdjson_result<double> document::get() && noexcept { return std::forward<document>(*this).get_double(); }
+template<> simdjson_deprecated simdjson_inline simdjson_result<float> document::get() && noexcept { return std::forward<document>(*this).get_float(); }
 template<> simdjson_deprecated simdjson_inline simdjson_result<uint64_t> document::get() && noexcept { return std::forward<document>(*this).get_uint64(); }
 template<> simdjson_deprecated simdjson_inline simdjson_result<int64_t> document::get() && noexcept { return std::forward<document>(*this).get_int64(); }
 template<> simdjson_deprecated simdjson_inline simdjson_result<bool> document::get() && noexcept { return std::forward<document>(*this).get_bool(); }
@@ -70112,6 +87412,22 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<arm64::ondemand::docume
   if (error()) { return error(); }
   return first.get_int32();
 }
+simdjson_inline simdjson_result<uint16_t> simdjson_result<arm64::ondemand::document>::get_uint16() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<arm64::ondemand::document>::get_int16() noexcept {
+  if (error()) { return error(); }
+  return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<arm64::ondemand::document>::get_uint8() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<arm64::ondemand::document>::get_int8() noexcept {
+  if (error()) { return error(); }
+  return first.get_int8();
+}
 simdjson_inline simdjson_result<double> simdjson_result<arm64::ondemand::document>::get_double() noexcept {
   if (error()) { return error(); }
   return first.get_double();
@@ -70120,10 +87436,36 @@ simdjson_inline simdjson_result<double> simdjson_result<arm64::ondemand::documen
   if (error()) { return error(); }
   return first.get_double_in_string();
 }
+simdjson_inline simdjson_result<float> simdjson_result<arm64::ondemand::document>::get_float() noexcept {
+  if (error()) { return error(); }
+  return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<arm64::ondemand::document>::get_float_in_string() noexcept {
+  if (error()) { return error(); }
+  return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<arm64::ondemand::document>::get_float32() noexcept {
+  if (error()) { return error(); }
+  return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<arm64::ondemand::document>::get_float64() noexcept {
+  if (error()) { return error(); }
+  return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<std::string_view> simdjson_result<arm64::ondemand::document>::get_string(bool allow_replacement) noexcept {
   if (error()) { return error(); }
   return first.get_string(allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<arm64::ondemand::document>::get_u8string(bool allow_replacement) noexcept {
+  if (error()) { return error(); }
+  return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code simdjson_result<arm64::ondemand::document>::get_string(string_type& receiver, bool allow_replacement) noexcept {
   if (error()) { return error(); }
@@ -70151,22 +87493,46 @@ simdjson_inline simdjson_result<bool> simdjson_result<arm64::ondemand::document>
 }

 template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<arm64::ondemand::document>::get() & noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<arm64::ondemand::document>::get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, arm64::ondemand::document>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>();
 }
 template<typename T>
-simdjson_deprecated simdjson_inline simdjson_result<T> simdjson_result<arm64::ondemand::document>::get() && noexcept {
+simdjson_deprecated simdjson_inline simdjson_result<T> simdjson_result<arm64::ondemand::document>::get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, arm64::ondemand::document>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return std::forward<arm64::ondemand::document>(first).get<T>();
 }
 template<typename T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<arm64::ondemand::document>::get(T &out) & noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<arm64::ondemand::document>::get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, arm64::ondemand::document>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>(out);
 }
 template<typename T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<arm64::ondemand::document>::get(T &out) && noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<arm64::ondemand::document>::get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, arm64::ondemand::document>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return std::forward<arm64::ondemand::document>(first).get<T>(out);
 }
@@ -70235,27 +87601,27 @@ simdjson_inline simdjson_result<arm64::ondemand::document>::operator arm64::onde
 }
 simdjson_inline simdjson_result<arm64::ondemand::document>::operator uint64_t() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_uint64();
 }
 simdjson_inline simdjson_result<arm64::ondemand::document>::operator int64_t() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_int64();
 }
 simdjson_inline simdjson_result<arm64::ondemand::document>::operator double() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_double();
 }
 simdjson_inline simdjson_result<arm64::ondemand::document>::operator std::string_view() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_string();
 }
 simdjson_inline simdjson_result<arm64::ondemand::document>::operator arm64::ondemand::raw_json_string() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_raw_json_string();
 }
 simdjson_inline simdjson_result<arm64::ondemand::document>::operator bool() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_bool();
 }
 simdjson_inline simdjson_result<arm64::ondemand::document>::operator arm64::ondemand::value() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
@@ -70345,21 +87711,38 @@ simdjson_inline simdjson_result<uint64_t> document_reference::get_uint64() noexc
 simdjson_inline simdjson_result<uint64_t> document_reference::get_uint64_in_string() noexcept { return doc->get_root_value_iterator().get_root_uint64_in_string(false); }
 simdjson_inline simdjson_result<int64_t> document_reference::get_int64() noexcept { return doc->get_root_value_iterator().get_root_int64(false); }
 simdjson_inline simdjson_result<int64_t> document_reference::get_int64_in_string() noexcept { return doc->get_root_value_iterator().get_root_int64_in_string(false); }
-simdjson_inline simdjson_result<uint32_t> document_reference::get_uint32() noexcept {
-  uint64_t result;
-  SIMDJSON_TRY(doc->get_root_value_iterator().get_root_uint64(false).get(result));
-  if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<uint32_t>(result);
-}
-simdjson_inline simdjson_result<int32_t> document_reference::get_int32() noexcept {
-  int64_t result;
-  SIMDJSON_TRY(doc->get_root_value_iterator().get_root_int64(false).get(result));
-  if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<int32_t>(result);
-}
+simdjson_inline simdjson_result<uint32_t> document_reference::get_uint32() noexcept { return narrow_integer<uint32_t>(get_uint64()); }
+simdjson_inline simdjson_result<int32_t> document_reference::get_int32() noexcept { return narrow_integer<int32_t>(get_int64()); }
+simdjson_inline simdjson_result<uint16_t> document_reference::get_uint16() noexcept { return narrow_integer<uint16_t>(get_uint64()); }
+simdjson_inline simdjson_result<int16_t> document_reference::get_int16() noexcept { return narrow_integer<int16_t>(get_int64()); }
+simdjson_inline simdjson_result<uint8_t> document_reference::get_uint8() noexcept { return narrow_integer<uint8_t>(get_uint64()); }
+simdjson_inline simdjson_result<int8_t> document_reference::get_int8() noexcept { return narrow_integer<int8_t>(get_int64()); }
 simdjson_inline simdjson_result<double> document_reference::get_double() noexcept { return doc->get_root_value_iterator().get_root_double(false); }
-simdjson_inline simdjson_result<double> document_reference::get_double_in_string() noexcept { return doc->get_root_value_iterator().get_root_double(false); }
+simdjson_inline simdjson_result<double> document_reference::get_double_in_string() noexcept { return doc->get_root_value_iterator().get_root_double_in_string(false); }
+simdjson_inline simdjson_result<float> document_reference::get_float() noexcept { return doc->get_root_value_iterator().get_root_float(false); }
+simdjson_inline simdjson_result<float> document_reference::get_float_in_string() noexcept { return doc->get_root_value_iterator().get_root_float_in_string(false); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> document_reference::get_float32() noexcept {
+  float result;
+  SIMDJSON_TRY(get_float().get(result));
+  return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> document_reference::get_float64() noexcept {
+  double result;
+  SIMDJSON_TRY(get_double().get(result));
+  return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<std::string_view> document_reference::get_string(bool allow_replacement) noexcept { return doc->get_root_value_iterator().get_root_string(false, allow_replacement); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> document_reference::get_u8string(bool allow_replacement) noexcept {
+  std::string_view content;
+  SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+  return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code document_reference::get_string(string_type& receiver, bool allow_replacement) noexcept { return doc->get_root_value_iterator().get_root_string(receiver, false, allow_replacement); }
 simdjson_inline simdjson_result<std::string_view> document_reference::get_wobbly_string() noexcept { return doc->get_root_value_iterator().get_root_wobbly_string(false); }
@@ -70371,11 +87754,25 @@ template<> simdjson_inline simdjson_result<array> document_reference::get() & no
 template<> simdjson_inline simdjson_result<object> document_reference::get() & noexcept { return get_object(); }
 template<> simdjson_inline simdjson_result<raw_json_string> document_reference::get() & noexcept { return get_raw_json_string(); }
 template<> simdjson_inline simdjson_result<std::string_view> document_reference::get() & noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> document_reference::get() & noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_inline simdjson_result<double> document_reference::get() & noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> document_reference::get() & noexcept { return get_float(); }
 template<> simdjson_inline simdjson_result<uint64_t> document_reference::get() & noexcept { return get_uint64(); }
 template<> simdjson_inline simdjson_result<int64_t> document_reference::get() & noexcept { return get_int64(); }
 template<> simdjson_inline simdjson_result<uint32_t> document_reference::get() & noexcept { return get_uint32(); }
 template<> simdjson_inline simdjson_result<int32_t> document_reference::get() & noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> document_reference::get() & noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> document_reference::get() & noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> document_reference::get() & noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> document_reference::get() & noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> document_reference::get() & noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> document_reference::get() & noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<> simdjson_inline simdjson_result<bool> document_reference::get() & noexcept { return get_bool(); }
 template<> simdjson_inline simdjson_result<value> document_reference::get() & noexcept { return get_value(); }
 #if SIMDJSON_EXCEPTIONS
@@ -70521,6 +87918,22 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<arm64::ondemand::docume
   if (error()) { return error(); }
   return first.get_int32();
 }
+simdjson_inline simdjson_result<uint16_t> simdjson_result<arm64::ondemand::document_reference>::get_uint16() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<arm64::ondemand::document_reference>::get_int16() noexcept {
+  if (error()) { return error(); }
+  return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<arm64::ondemand::document_reference>::get_uint8() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<arm64::ondemand::document_reference>::get_int8() noexcept {
+  if (error()) { return error(); }
+  return first.get_int8();
+}
 simdjson_inline simdjson_result<double> simdjson_result<arm64::ondemand::document_reference>::get_double() noexcept {
   if (error()) { return error(); }
   return first.get_double();
@@ -70529,10 +87942,36 @@ simdjson_inline simdjson_result<double> simdjson_result<arm64::ondemand::documen
   if (error()) { return error(); }
   return first.get_double_in_string();
 }
+simdjson_inline simdjson_result<float> simdjson_result<arm64::ondemand::document_reference>::get_float() noexcept {
+  if (error()) { return error(); }
+  return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<arm64::ondemand::document_reference>::get_float_in_string() noexcept {
+  if (error()) { return error(); }
+  return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<arm64::ondemand::document_reference>::get_float32() noexcept {
+  if (error()) { return error(); }
+  return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<arm64::ondemand::document_reference>::get_float64() noexcept {
+  if (error()) { return error(); }
+  return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<std::string_view> simdjson_result<arm64::ondemand::document_reference>::get_string(bool allow_replacement) noexcept {
   if (error()) { return error(); }
   return first.get_string(allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<arm64::ondemand::document_reference>::get_u8string(bool allow_replacement) noexcept {
+  if (error()) { return error(); }
+  return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code simdjson_result<arm64::ondemand::document_reference>::get_string(string_type& receiver, bool allow_replacement) noexcept {
   if (error()) { return error(); }
@@ -70559,22 +87998,46 @@ simdjson_inline simdjson_result<bool> simdjson_result<arm64::ondemand::document_
   return first.is_null();
 }
 template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<arm64::ondemand::document_reference>::get() & noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<arm64::ondemand::document_reference>::get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, arm64::ondemand::document_reference>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>();
 }
 template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<arm64::ondemand::document_reference>::get() && noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<arm64::ondemand::document_reference>::get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, arm64::ondemand::document_reference>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return std::forward<arm64::ondemand::document_reference>(first).get<T>();
 }
 template <class T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<arm64::ondemand::document_reference>::get(T &out) & noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<arm64::ondemand::document_reference>::get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, arm64::ondemand::document_reference>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>(out);
 }
 template <class T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<arm64::ondemand::document_reference>::get(T &out) && noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<arm64::ondemand::document_reference>::get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, arm64::ondemand::document_reference>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return std::forward<arm64::ondemand::document_reference>(first).get<T>(out);
 }
@@ -70636,27 +88099,27 @@ simdjson_inline simdjson_result<arm64::ondemand::document_reference>::operator a
 }
 simdjson_inline simdjson_result<arm64::ondemand::document_reference>::operator uint64_t() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_uint64();
 }
 simdjson_inline simdjson_result<arm64::ondemand::document_reference>::operator int64_t() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_int64();
 }
 simdjson_inline simdjson_result<arm64::ondemand::document_reference>::operator double() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_double();
 }
 simdjson_inline simdjson_result<arm64::ondemand::document_reference>::operator std::string_view() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_string();
 }
 simdjson_inline simdjson_result<arm64::ondemand::document_reference>::operator arm64::ondemand::raw_json_string() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_raw_json_string();
 }
 simdjson_inline simdjson_result<arm64::ondemand::document_reference>::operator bool() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_bool();
 }
 simdjson_inline simdjson_result<arm64::ondemand::document_reference>::operator arm64::ondemand::value() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
@@ -70722,6 +88185,7 @@ simdjson_warn_unused simdjson_inline error_code simdjson_result<arm64::ondemand:
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */

 #include <algorithm>
+#include <cstring>
 #include <stdexcept>

 namespace simdjson {
@@ -70808,23 +88272,20 @@ simdjson_inline document_stream::document_stream(
   const uint8_t *_buf,
   size_t _len,
   size_t _batch_size,
-  bool _allow_comma_separated
+  bool _allow_comma_separated,
+  stream_format _format
 ) noexcept
   : parser{&_parser},
     buf{_buf},
     len{_len},
     batch_size{_batch_size <= MINIMAL_BATCH_SIZE ? MINIMAL_BATCH_SIZE : _batch_size},
     allow_comma_separated{_allow_comma_separated},
+    format{_format},
     error{SUCCESS}
     #ifdef SIMDJSON_THREADS_ENABLED
     , use_thread(_parser.threaded) // we need to make a copy because _parser.threaded can change
     #endif
 {
-#ifdef SIMDJSON_THREADS_ENABLED
-  if(worker.get() == nullptr) {
-    error = MEMALLOC;
-  }
-#endif
 }

 simdjson_inline document_stream::document_stream() noexcept
@@ -70833,6 +88294,7 @@ simdjson_inline document_stream::document_stream() noexcept
     len{0},
     batch_size{0},
     allow_comma_separated{false},
+    format{stream_format::whitespace_delimited},
     error{UNINITIALIZED}
     #ifdef SIMDJSON_THREADS_ENABLED
     , use_thread(false)
@@ -70852,6 +88314,9 @@ inline size_t document_stream::size_in_bytes() const noexcept {
 }

 inline size_t document_stream::truncated_bytes() const noexcept {
+  // Stage 1 returns EMPTY on zero-length input before it writes the index
+  // sentinels read below, so they would still hold a previous stream's values.
+  if (len == 0) { return 0; }
   if(error == CAPACITY) { return len - batch_start; }
   return parser->implementation->structural_indexes[parser->implementation->n_structural_indexes] - parser->implementation->structural_indexes[parser->implementation->n_structural_indexes + 1];
 }
@@ -70932,13 +88397,20 @@ inline void document_stream::start() noexcept {
     error = run_stage1(*parser, batch_start);
   }
   if (error) { return; }
-  doc_index = batch_start;
+  // For json_sequence mode, structural_indexes[0] points to the actual JSON value
+  // after the RS delimiter and any following whitespace. For regular mode, it is
+  // the offset from batch_start to the first document in the batch.
+  doc_index = batch_start + parser->implementation->structural_indexes[0];
   doc = document(json_iterator(&buf[batch_start], parser));
   doc.iter._streaming = true;

   #ifdef SIMDJSON_THREADS_ENABLED
   if (use_thread && next_batch_start() < len) {
     // Kick off the first thread on next batch if needed
+    if (worker.get() == nullptr) {
+      worker.reset(new(std::nothrow) stage1_worker());
+      if (worker.get() == nullptr) { error = MEMALLOC; return; }
+    }
     error = stage1_thread_parser.allocate(batch_size);
     if (error) { return; }
     worker->start_thread();
@@ -71013,12 +88485,69 @@ inline void document_stream::next() noexcept {
        */

       if (error) { continue; } // If the error was EMPTY, we may want to load another batch.
-      doc_index = batch_start;
+      doc_index = batch_start + parser->implementation->structural_indexes[0];
     }
   }
 }

+simdjson_inline uint8_t document_stream::document_delimiter() const noexcept {
+  switch (format) {
+    case stream_format::newline_delimited: return '\n';
+    case stream_format::json_sequence: return 0x1E;
+    default: return 0;
+  }
+}
+
+simdjson_inline bool document_stream::skip_to_delimiter(uint8_t delimiter) noexcept {
+  const uint8_t *const base = &buf[batch_start];
+  const token_position pos = doc.iter.position();
+  const token_position end = doc.iter.end_position();
+  if (pos >= end) { return false; }
+  const size_t here = size_t(doc.iter.token.peek(pos) - base);
+  const size_t batch_len =
+      (len - batch_start < batch_size) ? len - batch_start : batch_size;
+  if (here >= batch_len) { return false; }
+  const uint8_t *const found = static_cast<const uint8_t *>(
+      std::memchr(base + here, delimiter, batch_len - here));
+  if (found == nullptr) { return false; }
+
+  const uint32_t boundary = uint32_t(found - base);
+  // The answer is near `pos`: the delimiter ends the current document, while
+  // `end` spans the whole batch. Gallop first so the cost follows the distance
+  // rather than the size of the batch.
+  token_position lo = pos;
+  size_t hop = 1;
+  while (lo + hop < end && lo[hop] < boundary) { lo += hop; hop <<= 1; }
+  token_position hi = (lo + hop < end) ? lo + hop : end;
+  while (lo < hi) {
+    const token_position mid = lo + ((hi - lo) >> 1);
+    if (*mid < boundary) { lo = mid + 1; } else { hi = mid; }
+  }
+  doc.iter.token.set_position(lo);
+  return true;
+}
+
 inline void document_stream::next_document() noexcept {
+  // A delimiter that cannot occur inside a document tells us where the current
+  // one ends, so we can jump there instead of walking every structural. Only
+  // valid while the iterator is still inside the document: a consumed document
+  // already sits on the next one's first token, and skip_child() returns at
+  // once for it.
+  //
+  // The jump does not structure-validate the unread remainder of the current
+  // document: under newline_delimited / json_sequence the next delimiter is
+  // assumed to be the true document boundary. Callers that leave depth() > 0
+  // while violating that contract (e.g. pretty multi-line JSON under
+  // newline_delimited) can mis-align following documents; use
+  // whitespace_delimited if unsure.
+  const uint8_t delimiter = document_delimiter();
+  if (delimiter != 0 && !error && doc.iter.depth() > 0 &&
+      skip_to_delimiter(delimiter)) {
+    doc.iter._depth = 1;
+    doc.iter._string_buf_loc = parser->string_buf.get();
+    doc.iter._root = doc.iter.position();
+    return;
+  }
   // Go to next place where depth=0 (document depth)
   error = doc.iter.skip_child(0);
   if (error) { return; }
@@ -71042,10 +88571,35 @@ inline error_code document_stream::run_stage1(ondemand::parser &p, size_t _batch
   // This code only updates the structural index in the parser, it does not update any json_iterator
   // instance.
   size_t remaining = len - _batch_start;
+  stage1_mode mode;
   if (remaining <= batch_size) {
-    return p.implementation->stage1(&buf[_batch_start], remaining, stage1_mode::streaming_final);
+    // Final batch
+    switch (format) {
+      case stream_format::json_sequence:
+        mode = stage1_mode::json_sequence_final;
+        break;
+      case stream_format::comma_delimited:
+        mode = stage1_mode::comma_delimited_final;
+        break;
+      default:
+        mode = stage1_mode::streaming_final;
+        break;
+    }
+    return p.implementation->stage1(&buf[_batch_start], remaining, mode);
   } else {
-    return p.implementation->stage1(&buf[_batch_start], batch_size, stage1_mode::streaming_partial);
+    // Partial batch
+    switch (format) {
+      case stream_format::json_sequence:
+        mode = stage1_mode::json_sequence_partial;
+        break;
+      case stream_format::comma_delimited:
+        mode = stage1_mode::comma_delimited_partial;
+        break;
+      default:
+        mode = stage1_mode::streaming_partial;
+        break;
+    }
+    return p.implementation->stage1(&buf[_batch_start], batch_size, mode);
   }
 }

@@ -71054,11 +88608,19 @@ simdjson_inline size_t document_stream::iterator::current_index() const noexcept
 }

 simdjson_inline std::string_view document_stream::iterator::source() const noexcept {
-  auto depth = stream->doc.iter.depth();
+  // On error (e.g., CAPACITY), there is no document to walk: return the rest of
+  // the input, as the DOM document_stream does.
+  if (stream->error) {
+    return std::string_view(reinterpret_cast<const char*>(stream->buf) + current_index(), stream->len - current_index());
+  }
+  // Always walk from the root of the document, whatever the current position
+  // of the document iterator: the user may have already consumed part of the
+  // document, so the iterator's current depth must not be used here.
+  depth_t depth = 1;
   auto cur_struct_index = stream->doc.iter._root - stream->parser->implementation->structural_indexes.get();

-  // If at root, process the first token to determine if scalar value
-  if (stream->doc.iter.at_root()) {
+  // Process the first token to determine if scalar value
+  {
     switch (stream->buf[stream->batch_start + stream->parser->implementation->structural_indexes[cur_struct_index]]) {
       case '{': case '[':   // Depth=1 already at start of document
         break;
@@ -71066,14 +88628,47 @@ simdjson_inline std::string_view document_stream::iterator::source() const noexc
         depth--;
         break;
       default:    // Scalar value document
-        // TODO: We could remove trailing whitespaces
         // This returns a string spanning from start of value to the beginning of the next document (excluded)
         {
           auto next_index = stream->batch_start + stream->parser->implementation->structural_indexes[++cur_struct_index];
           // normally the length would be next_index - current_index() - 1, except for the last document
           size_t svlen = next_index - current_index();
           const char *start = reinterpret_cast<const char*>(stream->buf) + current_index();
-          while(svlen > 1 && (std::isspace(start[svlen-1]) || start[svlen-1] == '\0')) {
+          // When the scalar is followed by a truncated document, the structural
+          // indexes of that document were dropped and next_index is the end of
+          // the input, so we bound the scalar by scanning the token itself.
+          size_t token_len = 0;
+          if (*start == '"') {
+            token_len = 1;
+            while (token_len < svlen) {
+              char c = start[token_len++];
+              if (c == '\\') {
+                token_len++;
+              } else if (c == '"') {
+                break;
+              }
+            }
+          } else {
+            while (token_len < svlen) {
+              char c = start[token_len];
+              if (std::isspace(static_cast<unsigned char>(c)) || c == ',' || c == '{' || c == '[' || c == '\0' || static_cast<uint8_t>(c) == 0x1E) {
+                break;
+              }
+              token_len++;
+            }
+          }
+          if (token_len > 0 && token_len < svlen) {
+            svlen = token_len;
+          }
+          // Trim trailing whitespace, NUL, and RS (0x1E). In RFC 7464
+          // json_sequence mode the scanner classifies RS as a scalar
+          // character, so an RS-prefixed scalar document (number / true /
+          // false / null / string) has no closing structural index and the
+          // slice runs all the way up to the next document's RS. RS cannot
+          // legally appear in a JSON value at the source level (control
+          // characters in strings must be escaped as \u001E), so stripping
+          // it is safe in every stream_format.
+          while(svlen > 1 && (std::isspace(static_cast<unsigned char>(start[svlen-1])) || start[svlen-1] == '\0' || static_cast<uint8_t>(start[svlen-1]) == 0x1E || (stream->format == stream_format::comma_delimited && start[svlen-1] == ','))) {
             svlen--;
           }
           return std::string_view(start, svlen);
@@ -71198,11 +88793,19 @@ simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> field::un
   return answer;
 }

+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_warn_unused simdjson_result<std::u8string_view> field::unescaped_u8key(bool allow_replacement) noexcept {
+  std::string_view key;
+  SIMDJSON_TRY( unescaped_key(allow_replacement).get(key) );
+  return internal::as_u8string_view(key);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 template <typename string_type>
 simdjson_inline simdjson_warn_unused error_code field::unescaped_key(string_type& receiver, bool allow_replacement) noexcept {
   std::string_view key;
   SIMDJSON_TRY( unescaped_key(allow_replacement).get(key) );
-  receiver = key;
+  internal::assign_utf8(receiver, key);
   return SUCCESS;
 }

@@ -71224,6 +88827,12 @@ simdjson_inline std::string_view field::escaped_key() const noexcept {
   return std::string_view(reinterpret_cast<const char*>(first.buf), end_quote - first.buf);
 }

+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline std::u8string_view field::escaped_u8key() const noexcept {
+  return internal::as_u8string_view(escaped_key());
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 simdjson_inline value &field::value() & noexcept {
   return second;
 }
@@ -71268,11 +88877,25 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<arm64::ondeman
   return first.escaped_key();
 }

+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<arm64::ondemand::field>::escaped_u8key() noexcept {
+  if (error()) { return error(); }
+  return first.escaped_u8key();
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 simdjson_inline simdjson_result<std::string_view> simdjson_result<arm64::ondemand::field>::unescaped_key(bool allow_replacement) noexcept {
   if (error()) { return error(); }
   return first.unescaped_key(allow_replacement);
 }

+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<arm64::ondemand::field>::unescaped_u8key(bool allow_replacement) noexcept {
+  if (error()) { return error(); }
+  return first.unescaped_u8key(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 template<typename string_type>
 simdjson_warn_unused simdjson_inline error_code simdjson_result<arm64::ondemand::field>::unescaped_key(string_type &receiver, bool allow_replacement) noexcept {
   if (error()) { return error(); }
@@ -71316,6 +88939,9 @@ simdjson_inline json_iterator::json_iterator(json_iterator &&other) noexcept
     _depth{other._depth},
     _root{other._root},
     _streaming{other._streaming}
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+    , _allow_incomplete_json{other._allow_incomplete_json}
+#endif
 {
   other.parser = nullptr;
 }
@@ -71327,6 +88953,9 @@ simdjson_inline json_iterator &json_iterator::operator=(json_iterator &&other) n
   _depth = other._depth;
   _root = other._root;
   _streaming = other._streaming;
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  _allow_incomplete_json = other._allow_incomplete_json;
+#endif
   other.parser = nullptr;
   return *this;
 }
@@ -71353,7 +88982,8 @@ simdjson_inline json_iterator::json_iterator(const uint8_t *buf, ondemand::parse
       _string_buf_loc{parser->string_buf.get()},
       _depth{1},
       _root{parser->implementation->structural_indexes.get()},
-      _streaming{streaming}
+      _streaming{streaming},
+      _allow_incomplete_json{true}

 {
   logger::log_headers();
@@ -71425,7 +89055,8 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::skip_child(depth_
 #endif // SIMDJSON_CHECK_EOF
       break;
     case '"':
-      if(*peek() == ':') {
+      // At the end, peek() would read the sentinel, which points into the padding.
+      if(!at_end() && *peek() == ':') {
         // We are at a key!!!
         // This might happen if you just started an object and you skip it immediately.
         // Performance note: it would be nice to get rid of this check as it is somewhat
@@ -71468,7 +89099,7 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::skip_child(depth_
     }
   }

-  return report_error(TAPE_ERROR, "not enough close braces");
+  return report_error(INCOMPLETE_ARRAY_OR_OBJECT, "not enough close braces");
 }

 SIMDJSON_POP_DISABLE_WARNINGS
@@ -71485,6 +89116,17 @@ simdjson_inline bool json_iterator::streaming() const noexcept {
   return _streaming;
 }

+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+simdjson_inline bool json_iterator::allow_incomplete_json() const noexcept {
+  return _allow_incomplete_json;
+}
+
+simdjson_inline size_t json_iterator::remaining_input_length(const uint8_t *json) const noexcept {
+  const uint8_t *end = token.buf + parser->_document_len;
+  return json < end ? size_t(end - json) : 0;
+}
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+
 simdjson_inline token_position json_iterator::root_position() const noexcept {
   return _root;
 }
@@ -71767,7 +89409,7 @@ inline std::ostream& operator<<(std::ostream& out, json_type type) noexcept {
         case json_type::string: out << "string"; break;
         case json_type::boolean: out << "boolean"; break;
         case json_type::null: out << "null"; break;
-        default: SIMDJSON_UNREACHABLE();
+        case json_type::unknown: out << "unknown"; break;
     }
     return out;
 }
@@ -72106,6 +89748,10 @@ inline void log_line(const json_iterator &iter, token_position index, depth_t de
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_iterator.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value-inl.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/jsonpathutil.h" */
+/* amalgamation skipped (editor-only): #include <utility> */
+/* amalgamation skipped (editor-only): #if SIMDJSON_SUPPORTS_CONCEPTS */
+/* amalgamation skipped (editor-only): #include <tuple> // std::forward_as_tuple/get for the variadic for_each adapter */
+/* amalgamation skipped (editor-only): #endif */
 /* amalgamation skipped (editor-only): #if SIMDJSON_STATIC_REFLECTION */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_string_builder.h"  // for constevalutil::fixed_string */
 /* amalgamation skipped (editor-only): #include <meta> */
@@ -72135,12 +89781,21 @@ simdjson_inline simdjson_result<value> object::find_field_unordered(const std::s
   return value(iter.child());
 }
 simdjson_inline simdjson_result<value> object::operator[](const std::string_view key) & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   return find_field_unordered(key);
 }
 simdjson_inline simdjson_result<value> object::operator[](const std::string_view key) && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   return std::forward<object>(*this).find_field_unordered(key);
 }
 simdjson_inline simdjson_result<value> object::find_field(const std::string_view key) & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   bool has_value;
   SIMDJSON_TRY( iter.find_field_raw(key).get(has_value) );
   if (!has_value) {
@@ -72150,6 +89805,9 @@ simdjson_inline simdjson_result<value> object::find_field(const std::string_view
   return value(iter.child());
 }
 simdjson_inline simdjson_result<value> object::find_field(const std::string_view key) && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   bool has_value;
   SIMDJSON_TRY( iter.find_field_raw(key).get(has_value) );
   if (!has_value) {
@@ -72159,6 +89817,150 @@ simdjson_inline simdjson_result<value> object::find_field(const std::string_view
   return value(iter.child());
 }

+#if SIMDJSON_SUPPORTS_CONCEPTS
+template <typename Selector, typename Func>
+  requires key_selector_type<Selector> &&
+           std::is_invocable_v<Func&, std::size_t, value>
+simdjson_flatten simdjson_inline for_each_result object::for_each(Func&& on_match)
+    noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, value>) {
+  // Single pass driven directly by the value_iterator, mirroring
+  // find_field_unordered_raw + value(iter.child()). Compared to walking via
+  // object_iterator/field, this avoids constructing a simdjson_result<field> and
+  // a field (key + value) for every field -- and the development-check bookkeeping
+  // in object_iterator -- building a value only for the fields that actually match.
+  // We operate on a copy of the iterator, as object::begin() would.
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  // Mirror object::begin(): for_each must start at the beginning of the object,
+  // not from some position left behind by a prior find_field on the same object.
+  if (!iter.is_at_iterator_start()) { return {OUT_OF_ORDER_ITERATION, 0}; }
+#endif
+  value_iterator it = iter;
+  std::size_t matched = 0;
+  // Track which selector indices have already matched, as a compile-time bitset
+  // (one 64-bit word per 64 keys): the callback fires once per key, on its first
+  // occurrence, and we stop as soon as every key has matched.
+  constexpr std::size_t seen_words = (Selector::size() + 63) / 64;
+  std::array<std::uint64_t, seen_words> seen{};
+  while (it.is_open()) {
+    raw_json_string key;
+    error_code error;
+    std::size_t idx;
+    if constexpr (Selector::window.ok) {
+      // A window selector confirms a key from its raw bytes alone (the closing
+      // quote bounds it), so we take the length-free path: field_key (no backward
+      // scan, mirroring the ordered find_field) feeding match_raw(raw_json_string).
+      if ((error = it.field_key().get(key))) { it.abandon(); return {error, matched}; }
+      if ((error = it.field_value())) { it.abandon(); return {error, matched}; }
+      idx = Selector::match_raw(key);
+    } else {
+      // Otherwise derive the key length from the structural index (the following
+      // ':' token) to feed the length-aware hash, avoiding a forward SIMD scan.
+      std::size_t key_len;
+      if ((error = it.field_key_with_length(key, key_len))) { it.abandon(); return {error, matched}; }
+      if ((error = it.field_value())) { it.abandon(); return {error, matched}; }
+      idx = Selector::match_raw(key.raw(), key_len);
+    }
+    if (idx < Selector::size()) {
+      const std::uint64_t seen_bit = std::uint64_t{1} << (idx & 63);
+      std::uint64_t &seen_word = seen[idx >> 6];
+      if (!(seen_word & seen_bit)) {
+        seen_word |= seen_bit;
+        value matched_value(it.child());
+        // The callback may return void or anything convertible to error_code
+        // (error_code itself, or a for_each_result from a nested for_each). When
+        // it yields an error_code, we stop at the first non-SUCCESS result and
+        // propagate it so the caller can surface value-parse errors (for example,
+        // a type mismatch on a matched field). A void-returning callback is
+        // responsible for handling its own errors.
+        if constexpr (std::is_convertible_v<decltype(on_match(idx, matched_value)), error_code>) {
+          // Unlike the internal-error paths above, a callback error does not
+          // abandon the iterator: we leave it recoverable so the caller can keep
+          // using the object (or its parent) after handling the error.
+          if ((error = on_match(idx, matched_value))) { return {error, matched}; }
+        } else {
+          on_match(idx, matched_value);
+        }
+        if (++matched >= Selector::size()) { break; }
+      }
+    }
+    // Mirror object_iterator::operator++'s safety rail: if the callback consumed
+    // the value and left the iterator closed or in error (e.g. a void callback
+    // that swallowed a fatal sub-iteration error), stop here rather than calling
+    // skip_child on a closed iterator.
+    if (!it.is_open()) { break; }
+    // Skip the value (a no-op if the callback consumed it) and step to the next
+    // field; has_next_field() ends the container on '}', which closes the loop.
+    if ((error = it.skip_child())) { it.abandon(); return {error, matched}; }
+    if ((error = it.has_next_field().error())) { return {error, matched}; }
+  }
+  return {SUCCESS, matched};
+}
+
+namespace key_selector_for_each_detail {
+
+// Dispatch a matched value to the I-th handler in the tuple (0-based). A handler
+// is either an invocable taking the value (void- or error_code-returning) or a
+// deserialization target, in which case we assign via value::get. We always
+// return error_code so the core (index, value) for_each can treat the adapter
+// uniformly.
+template <typename Tuple, std::size_t... Is>
+simdjson_really_inline error_code dispatch_value(
+    std::size_t idx, Tuple& handlers, value v, std::index_sequence<Is...>) {
+  error_code err = SUCCESS;
+  auto try_one = [&](auto Ic) {
+    constexpr std::size_t I = decltype(Ic)::value;
+    if (idx == I) {
+      auto&& h = std::get<I>(handlers);
+      using H = std::remove_reference_t<decltype(h)>;
+      if constexpr (std::is_invocable_v<H&, value>) {
+        // A handler returning void runs for its side effects; one returning
+        // anything convertible to error_code (error_code, or a for_each_result
+        // from a nested for_each) has its error captured and propagated.
+        if constexpr (std::is_convertible_v<decltype(h(v)), error_code>) {
+          err = h(v);
+        } else {
+          h(v);
+        }
+      } else {
+        // Direct deserialization target: assign the matched value into it.
+        err = v.get(h);
+      }
+    }
+  };
+  (try_one(std::integral_constant<std::size_t, Is>{}), ...);
+  return err;
+}
+
+} // namespace key_selector_for_each_detail
+
+template <typename Selector, typename... Handlers>
+  requires key_selector_type<Selector> &&
+           (sizeof...(Handlers) == Selector::size()) &&
+           (key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_flatten simdjson_inline for_each_result object::for_each(Handlers&&... on_match)
+    noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+  auto handlers = std::forward_as_tuple(std::forward<Handlers>(on_match)...);
+  // Reuse the single (index, value) implementation via a tiny adapter.
+  // The adapter is called once per *matched* key (very few); the hot path
+  // (iteration + match_raw + seen bitset) stays exactly the same.
+  return this->template for_each<Selector>(
+      [&](std::size_t i, value v) -> error_code {
+        return key_selector_for_each_detail::dispatch_value(
+            i, handlers, v, std::make_index_sequence<Selector::size()>{});
+      });
+}
+
+template <constevalutil::fixed_string... Keys, typename... Handlers>
+  requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+           (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+           (key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline for_each_result object::for_each(Handlers&&... on_match)
+    noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+  using Selector = key_selector<Keys...>;
+  return this->template for_each<Selector>(std::forward<Handlers>(on_match)...);
+}
+#endif
+
 simdjson_inline simdjson_result<object> object::start(value_iterator &iter) noexcept {
   SIMDJSON_TRY( iter.start_object().error() );
   return object(iter);
@@ -72194,6 +89996,9 @@ simdjson_warn_unused simdjson_inline error_code object::consume() noexcept {
 }

 simdjson_inline simdjson_result<std::string_view> object::raw_json() noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   const uint8_t * starting_point{iter.peek_start()};
   auto error = consume();
   if(error) { return error; }
@@ -72215,9 +90020,17 @@ simdjson_inline object::object(const value_iterator &_iter) noexcept
 {
 }

-simdjson_inline simdjson_result<object_iterator> object::begin() noexcept {
+simdjson_inline simdjson_result<object_iterator> object::begin() & noexcept {
 #if SIMDJSON_DEVELOPMENT_CHECKS
-  if (!iter.is_at_iterator_start()) { return OUT_OF_ORDER_ITERATION; }
+  if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
+  return object_iterator(iter, this);
+#endif
+  return object_iterator(iter);
+}
+simdjson_inline simdjson_result<object_iterator> object::begin() && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  // The object is a temporary that the iterator may outlive: do not lock it.
+  if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
 #endif
   return object_iterator(iter);
 }
@@ -72226,7 +90039,9 @@ simdjson_inline simdjson_result<object_iterator> object::end() noexcept {
 }

 inline simdjson_result<value> object::at_pointer(std::string_view json_pointer) noexcept {
-  if (json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
+  // An empty pointer has no json_pointer[0]: with a default-constructed
+  // std::string_view, reading it dereferences a null pointer.
+  if (json_pointer.empty() || json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
   json_pointer = json_pointer.substr(1);
   size_t slash = json_pointer.find('/');
   std::string_view key = json_pointer.substr(0, slash);
@@ -72328,6 +90143,9 @@ simdjson_inline simdjson_result<size_t> object::count_fields() & noexcept {
 }

 simdjson_inline simdjson_result<bool> object::is_empty() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   bool is_not_empty;
   auto error = iter.reset_object().get(is_not_empty);
   if(error) { return error; }
@@ -72335,9 +90153,41 @@ simdjson_inline simdjson_result<bool> object::is_empty() & noexcept {
 }

 simdjson_inline simdjson_result<bool> object::reset() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   return iter.reset_object();
 }

+simdjson_inline object_position object::get_current_position() const noexcept {
+  return object_position{iter.position(), iter.json_iter().depth()};
+}
+
+simdjson_inline error_code object::revert_position(object_position position) noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
+  // json_iterator::reenter_child() requires the live depth to be exactly
+  // one level shallower than the target (matching how every other depth
+  // transition in this iterator works), and under SIMDJSON_DEVELOPMENT_CHECKS
+  // additionally validates against the parser's per-depth container-start
+  // bookkeeping. Neither applies here: depending on what was captured and
+  // what has happened since (a scalar field fully consumed, a compound
+  // value left open, a find_field() miss that scanned past everything),
+  // the live depth when reverting can be any number of levels away from
+  // the captured one, and the captured depth is not necessarily a
+  // container's own start. reenter_at() moves directly, matching how
+  // reset_object() itself repositions without going through reenter_child().
+  iter.reenter_at(position.position, position.depth);
+  return SUCCESS;
+}
+
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline void object::set_locked(bool _locked) noexcept {
+  locked = _locked;
+}
+#endif
+
 #if SIMDJSON_SUPPORTS_CONCEPTS && SIMDJSON_STATIC_REFLECTION

 template<constevalutil::fixed_string... FieldNames, typename T>
@@ -72395,10 +90245,14 @@ simdjson_inline simdjson_result<arm64::ondemand::object>::simdjson_result(arm64:
 simdjson_inline simdjson_result<arm64::ondemand::object>::simdjson_result(error_code error) noexcept
     : implementation_simdjson_result_base<arm64::ondemand::object>(error) {}

-simdjson_inline simdjson_result<arm64::ondemand::object_iterator> simdjson_result<arm64::ondemand::object>::begin() noexcept {
+simdjson_inline simdjson_result<arm64::ondemand::object_iterator> simdjson_result<arm64::ondemand::object>::begin() & noexcept {
   if (error()) { return error(); }
   return first.begin();
 }
+simdjson_inline simdjson_result<arm64::ondemand::object_iterator> simdjson_result<arm64::ondemand::object>::begin() && noexcept {
+  if (error()) { return error(); }
+  return std::move(first).begin();
+}
 simdjson_inline simdjson_result<arm64::ondemand::object_iterator> simdjson_result<arm64::ondemand::object>::end() noexcept {
   if (error()) { return error(); }
   return first.end();
@@ -72452,11 +90306,55 @@ simdjson_inline error_code simdjson_result<arm64::ondemand::object>::for_each_at
   return first.for_each_at_path_with_wildcard(json_path, std::forward<Func>(callback));
 }

+#if SIMDJSON_SUPPORTS_CONCEPTS
+template <typename Selector, typename Func>
+  requires arm64::ondemand::key_selector_type<Selector> &&
+           std::is_invocable_v<Func&, std::size_t, arm64::ondemand::value>
+simdjson_inline arm64::ondemand::for_each_result
+simdjson_result<arm64::ondemand::object>::for_each(Func&& on_match)
+    noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, arm64::ondemand::value>) {
+  if (error()) { return {error(), 0}; }
+  return first.template for_each<Selector>(std::forward<Func>(on_match));
+}
+
+template <typename Selector, typename... Handlers>
+  requires arm64::ondemand::key_selector_type<Selector> &&
+           (sizeof...(Handlers) == Selector::size()) &&
+           (arm64::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline arm64::ondemand::for_each_result
+simdjson_result<arm64::ondemand::object>::for_each(Handlers&&... on_match)
+    noexcept(arm64::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+  if (error()) { return {error(), 0}; }
+  return first.template for_each<Selector>(std::forward<Handlers>(on_match)...);
+}
+
+template <constevalutil::fixed_string... Keys, typename... Handlers>
+  requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+           (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+           (arm64::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline arm64::ondemand::for_each_result
+simdjson_result<arm64::ondemand::object>::for_each(Handlers&&... on_match)
+    noexcept(arm64::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+  if (error()) { return {error(), 0}; }
+  return first.template for_each<Keys...>(std::forward<Handlers>(on_match)...);
+}
+#endif
+
 inline simdjson_result<bool> simdjson_result<arm64::ondemand::object>::reset() noexcept {
   if (error()) { return error(); }
   return first.reset();
 }

+inline simdjson_result<arm64::ondemand::object_position> simdjson_result<arm64::ondemand::object>::get_current_position() noexcept {
+  if (error()) { return error(); }
+  return first.get_current_position();
+}
+
+inline error_code simdjson_result<arm64::ondemand::object>::revert_position(arm64::ondemand::object_position position) noexcept {
+  if (error()) { return error(); }
+  return first.revert_position(position);
+}
+
 inline simdjson_result<bool> simdjson_result<arm64::ondemand::object>::is_empty() noexcept {
   if (error()) { return error(); }
   return first.is_empty();
@@ -72500,6 +90398,61 @@ simdjson_inline object_iterator::object_iterator(const value_iterator &_iter) no
   : iter{_iter}
 {}

+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline object_iterator::object_iterator(const value_iterator &_iter, object* _parent) noexcept
+  : parent{_parent}, iter{_iter}
+{
+  if (parent) parent->set_locked(true);
+}
+#endif
+
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline object_iterator::~object_iterator() noexcept
+{
+  if (parent) parent->set_locked(false);
+}
+
+simdjson_inline object_iterator::object_iterator(object_iterator&& other) noexcept
+  : has_been_referenced{other.has_been_referenced},
+    parent{other.parent},
+    iter{std::move(other.iter)}
+{
+  other.parent = nullptr;
+}
+
+simdjson_inline object_iterator& object_iterator::operator=(object_iterator&& other) noexcept {
+  if (this != &other)
+  {
+    if (parent)
+      parent->set_locked(false);
+    has_been_referenced = other.has_been_referenced;
+    parent = other.parent;
+    iter = std::move(other.iter);
+
+    other.parent = nullptr;
+  }
+  return *this;
+}
+
+simdjson_inline object_iterator::object_iterator(const object_iterator& other) noexcept
+  : has_been_referenced{other.has_been_referenced},
+    parent{nullptr},
+    iter{other.iter}
+{}
+
+simdjson_inline object_iterator& object_iterator::operator=(const object_iterator& other) noexcept {
+  if (this != &other)
+  {
+    if (parent)
+      parent->set_locked(false);
+    has_been_referenced = other.has_been_referenced;
+    parent = nullptr;
+    iter = other.iter;
+  }
+  return *this;
+}
+#endif
+
 simdjson_inline simdjson_result<field> object_iterator::operator*() noexcept {
 #if SIMDJSON_DEVELOPMENT_CHECKS
   // We must call * once per iteration.
@@ -72627,6 +90580,147 @@ simdjson_inline simdjson_result<arm64::ondemand::object_iterator> &simdjson_resu

 #endif // SIMDJSON_GENERIC_ONDEMAND_OBJECT_ITERATOR_INL_H
 /* end file simdjson/generic/ondemand/object_iterator-inl.h for arm64 */
+/* including simdjson/generic/ondemand/ranges-inl.h for arm64: #include "simdjson/generic/ondemand/ranges-inl.h" */
+/* begin file simdjson/generic/ondemand/ranges-inl.h for arm64 */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/ranges.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array_iterator-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object_iterator-inl.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#if SIMDJSON_SUPPORTS_RANGES
+
+namespace simdjson {
+namespace arm64 {
+namespace ondemand {
+
+//
+// array_range_iterator
+//
+
+simdjson_inline array_range_iterator::array_range_iterator(array_iterator iter) noexcept
+  : iter_{iter} {}
+
+simdjson_inline simdjson_result<value> array_range_iterator::operator*() const noexcept {
+  return *iter_;
+}
+
+simdjson_inline array_range_iterator& array_range_iterator::operator++() noexcept {
+  ++iter_;
+  return *this;
+}
+
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+simdjson_inline void array_range_iterator::operator++(int) noexcept {
+  ++*this;
+}
+SIMDJSON_POP_DISABLE_WARNINGS
+
+//
+// array_range
+//
+
+simdjson_inline array_range::array_range(array& arr) noexcept {
+  auto b = arr.begin();
+  if (b.error()) { error_ = b.error(); return; }
+  begin_ = b.value_unsafe();
+  end_ = arr.end().value_unsafe();
+}
+
+simdjson_inline array_range_iterator array_range::begin() noexcept {
+  return array_range_iterator(begin_);
+}
+
+simdjson_inline array_range_iterator array_range::end() noexcept {
+  return array_range_iterator(end_);
+}
+
+//
+// object_range_iterator
+//
+
+simdjson_inline object_range_iterator::object_range_iterator(object_iterator iter) noexcept
+  : iter_{iter} {}
+
+simdjson_inline simdjson_result<field> object_range_iterator::operator*() const noexcept {
+  return *iter_;
+}
+
+simdjson_inline object_range_iterator& object_range_iterator::operator++() noexcept {
+  ++iter_;
+  return *this;
+}
+
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+simdjson_inline void object_range_iterator::operator++(int) noexcept {
+  ++*this;
+}
+SIMDJSON_POP_DISABLE_WARNINGS
+
+
+//
+// object_range
+//
+
+simdjson_inline object_range::object_range(object& obj) noexcept {
+  auto b = obj.begin();
+  if (b.error()) { error_ = b.error(); return; }
+  begin_ = b.value_unsafe();
+  end_ = obj.end().value_unsafe();
+}
+
+simdjson_inline object_range_iterator object_range::begin() noexcept {
+  return object_range_iterator(begin_);
+}
+
+simdjson_inline object_range_iterator object_range::end() noexcept {
+  return object_range_iterator(end_);
+}
+
+//
+// Free functions
+//
+
+simdjson_inline array_range get_range(array& arr) noexcept {
+  return array_range(arr);
+}
+
+simdjson_inline object_range get_key_value_range(object& obj) noexcept {
+  return object_range(obj);
+}
+
+#if SIMDJSON_EXCEPTIONS
+simdjson_inline array_range get_range(simdjson_result<array> result) {
+  return array_range(result.value());
+}
+
+simdjson_inline object_range get_key_value_range(simdjson_result<object> result) {
+  return object_range(result.value());
+}
+#endif // SIMDJSON_EXCEPTIONS
+
+} // namespace ondemand
+} // namespace arm64
+} // namespace simdjson
+
+// Verify the range wrapper types satisfy the expected C++20 concepts.
+static_assert(std::input_iterator<simdjson::arm64::ondemand::array_range_iterator>);
+static_assert(std::input_iterator<simdjson::arm64::ondemand::object_range_iterator>);
+static_assert(std::ranges::input_range<simdjson::arm64::ondemand::array_range>);
+static_assert(std::ranges::input_range<simdjson::arm64::ondemand::object_range>);
+static_assert(std::ranges::view<simdjson::arm64::ondemand::array_range>);
+static_assert(std::ranges::view<simdjson::arm64::ondemand::object_range>);
+
+#endif // SIMDJSON_SUPPORTS_RANGES
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H
+/* end file simdjson/generic/ondemand/ranges-inl.h for arm64 */
 /* including simdjson/generic/ondemand/parser-inl.h for arm64: #include "simdjson/generic/ondemand/parser-inl.h" */
 /* begin file simdjson/generic/ondemand/parser-inl.h for arm64 */
 #ifndef SIMDJSON_GENERIC_ONDEMAND_PARSER_INL_H
@@ -72658,7 +90752,10 @@ simdjson_warn_unused simdjson_inline error_code parser::allocate(size_t new_capa

   // string_capacity copied from document::allocate
   _capacity = 0;
-  size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * new_capacity / 3 + SIMDJSON_PADDING, 64);
+  if(5 * (new_capacity / 3) + SIMDJSON_PADDING < SIMDJSON_PADDING) {
+    return CAPACITY; // overflow, only happen on legacy 32-bit systems with very large capacity
+  }
+  size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * (new_capacity / 3) + SIMDJSON_PADDING, 64);
   string_buf.reset(new (std::nothrow) uint8_t[string_capacity]);
 #if SIMDJSON_DEVELOPMENT_CHECKS
   start_positions.reset(new (std::nothrow) token_position[new_max_depth]);
@@ -72683,6 +90780,7 @@ simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate(p
   if (!json.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }

   json.remove_utf8_bom();
+  _document_len = json.length();

   // Allocate if needed
   if (capacity() < json.length() || !string_buf) {
@@ -72699,6 +90797,7 @@ simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate_a
   if (!json.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }

   json.remove_utf8_bom();
+  _document_len = json.length();

   // Allocate if needed
   if (capacity() < json.length() || !string_buf) {
@@ -72764,6 +90863,34 @@ simdjson_warn_unused simdjson_inline simdjson_result<json_iterator> parser::iter
   return json_iterator(reinterpret_cast<const uint8_t *>(json.data()), this);
 }

+inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size) noexcept {
+  if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
+  if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+    buf += 3;
+    len -= 3;
+  }
+  return document_stream(*this, buf, len, batch_size, false, stream_format::whitespace_delimited);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size) noexcept {
+  return iterate_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(padded_string_view s, size_t batch_size) noexcept {
+  if (!s.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
+  return iterate_many(s.data(), s.length(), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const padded_string &s, size_t batch_size) noexcept {
+  return iterate_many(padded_string_view(s), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const std::string &s, size_t batch_size) noexcept {
+  return iterate_many(padded_string_view(s), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size) noexcept {
+  return iterate_many(pad(s), batch_size);
+}
+
+#ifndef SIMDJSON_DISABLE_DEPRECATED_API
+SIMDJSON_PUSH_DISABLE_WARNINGS
+SIMDJSON_DISABLE_DEPRECATED_WARNING
 inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept {
   // Warning: no check is done on the buffer padding. We trust the user.
   if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
@@ -72771,8 +90898,11 @@ inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf,
     buf += 3;
     len -= 3;
   }
-  if(allow_comma_separated && batch_size < len) { batch_size = len; }
-  return document_stream(*this, buf, len, batch_size, allow_comma_separated);
+  // Map allow_comma_separated to stream_format::comma_delimited
+  if (allow_comma_separated) {
+    return document_stream(*this, buf, len, batch_size, false, stream_format::comma_delimited);
+  }
+  return document_stream(*this, buf, len, batch_size, false, stream_format::whitespace_delimited);
 }

 inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept {
@@ -72792,6 +90922,51 @@ inline simdjson_result<document_stream> parser::iterate_many(const std::string &
 inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size, bool allow_comma_separated) noexcept {
   return iterate_many(pad(s), batch_size, allow_comma_separated);
 }
+SIMDJSON_POP_DISABLE_WARNINGS
+#endif // SIMDJSON_DISABLE_DEPRECATED_API
+
+inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+  if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
+  if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+    buf += 3;
+    len -= 3;
+  }
+  if (format == stream_format::comma_delimited_array) {
+    // Strip leading JSON whitespace.
+    while (len > 0 && (buf[0] == ' ' || buf[0] == '\t' || buf[0] == '\n' || buf[0] == '\r')) {
+      buf++; len--;
+    }
+    // Expect the opening '['.
+    if (len == 0 || buf[0] != '[') { return TAPE_ERROR; }
+    buf++; len--;
+    // Strip trailing JSON whitespace.
+    while (len > 0 && (buf[len-1] == ' ' || buf[len-1] == '\t' || buf[len-1] == '\n' || buf[len-1] == '\r')) {
+      len--;
+    }
+    // Expect the closing ']'.
+    if (len == 0 || buf[len-1] != ']') { return TAPE_ERROR; }
+    len--;
+    // Fall through to comma_delimited over the array contents.
+    format = stream_format::comma_delimited;
+  }
+  return document_stream(*this, buf, len, batch_size, false, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+  return iterate_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(padded_string_view s, size_t batch_size, stream_format format) noexcept {
+  if (!s.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
+  return iterate_many(s.data(), s.length(), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const std::string &s, size_t batch_size, stream_format format) noexcept {
+  return iterate_many(padded_string_view(s), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size, stream_format format) noexcept {
+  return iterate_many(pad(s), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const padded_string &s, size_t batch_size, stream_format format) noexcept {
+  return iterate_many(padded_string_view(s), batch_size, format);
+}
 simdjson_pure simdjson_inline size_t parser::capacity() const noexcept {
   return _capacity;
 }
@@ -73199,6 +91374,27 @@ namespace simdjson {
 namespace arm64 {
 namespace ondemand {

+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+simdjson_inline bool raw_json_string_is_quote_terminated(const uint8_t *json, uint32_t max_len) noexcept {
+  bool escaping{false};
+  for (uint32_t i = 1; i < max_len; i++) {
+    switch (json[i]) {
+      case '"':
+        if (!escaping) { return true; }
+        escaping = false;
+        break;
+      case '\\':
+        escaping = !escaping;
+        break;
+      default:
+        escaping = false;
+        break;
+    }
+  }
+  return false;
+}
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+
 simdjson_inline value_iterator::value_iterator(
   json_iterator *json_iter,
   depth_t depth,
@@ -73586,6 +91782,22 @@ simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iter
   return raw_json_string(key);
 }

+simdjson_warn_unused simdjson_inline error_code value_iterator::field_key_with_length(raw_json_string &key, std::size_t &len) noexcept {
+  assert_at_next();
+
+  const uint8_t *k = _json_iter->return_current_and_advance();
+  if (*(k++) != '"') { return report_error(TAPE_ERROR, "Object key is not a string"); }
+  // After return_current_and_advance(), the current token is the ':' that follows
+  // the key. The closing quote sits just before it (only JSON whitespace may
+  // intervene), so step back from the ':' to the closing quote to get the length.
+  // In minified JSON this is a single back-step.
+  const char *q = reinterpret_cast<const char *>(_json_iter->peek());
+  do { --q; } while (*q != '"');
+  key = raw_json_string(k);
+  len = static_cast<std::size_t>(q - reinterpret_cast<const char *>(k));
+  return SUCCESS;
+}
+
 simdjson_warn_unused simdjson_inline error_code value_iterator::field_value() noexcept {
   assert_at_next();

@@ -73703,7 +91915,7 @@ simdjson_warn_unused simdjson_inline error_code value_iterator::get_string(strin
   auto saved_string_buf_loc = _json_iter->string_buf_loc();
   auto err = get_string(allow_replacement).get(content);
   if (err) { return err; }
-  receiver = content;
+  internal::assign_utf8(receiver, content);
   // Restore the string buffer location, effectively discarding any temporary string storage
   _json_iter->string_buf_loc() = saved_string_buf_loc;
   return SUCCESS;
@@ -73714,6 +91926,15 @@ simdjson_warn_unused simdjson_inline simdjson_result<std::string_view> value_ite
 simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iterator::get_raw_json_string() noexcept {
   auto json = peek_scalar("string");
   if (*json != '"') { return incorrect_type_error("Not a string"); }
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  if (_json_iter->allow_incomplete_json()) {
+    const size_t remaining_input_length = _json_iter->remaining_input_length(json);
+    const uint32_t max_len = remaining_input_length < peek_start_length() ? uint32_t(remaining_input_length) : peek_start_length();
+    if (!raw_json_string_is_quote_terminated(json, max_len)) {
+      return STRING_ERROR;
+    }
+  }
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
   advance_scalar("string");
   return raw_json_string(json+1);
 }
@@ -73747,6 +91968,16 @@ simdjson_warn_unused simdjson_inline simdjson_result<double> value_iterator::get
   if(result.error() == SUCCESS) { advance_non_root_scalar("double"); }
   return result;
 }
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_float() noexcept {
+  auto result = numberparsing::parse_float(peek_non_root_scalar("float"));
+  if(result.error() == SUCCESS) { advance_non_root_scalar("float"); }
+  return result;
+}
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_float_in_string() noexcept {
+  auto result = numberparsing::parse_float_in_string(peek_non_root_scalar("float"));
+  if(result.error() == SUCCESS) { advance_non_root_scalar("float"); }
+  return result;
+}
 simdjson_warn_unused simdjson_inline simdjson_result<bool> value_iterator::get_bool() noexcept {
   auto result = parse_bool(peek_non_root_scalar("bool"));
   if(result.error() == SUCCESS) { advance_non_root_scalar("bool"); }
@@ -73849,7 +92080,7 @@ simdjson_warn_unused simdjson_inline error_code value_iterator::get_root_string(
   auto saved_string_buf_loc = _json_iter->string_buf_loc();
   auto err = get_root_string(check_trailing, allow_replacement).get(content);
   if (err) { return err; }
-  receiver = content;
+  internal::assign_utf8(receiver, content);
   // Restore the string buffer location, effectively discarding any temporary string storage
   _json_iter->string_buf_loc() = saved_string_buf_loc;
   return SUCCESS;
@@ -73861,6 +92092,15 @@ simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iter
   auto json = peek_scalar("string");
   if (*json != '"') { return incorrect_type_error("Not a string"); }
   if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  if (_json_iter->allow_incomplete_json()) {
+    const size_t remaining_input_length = _json_iter->remaining_input_length(json);
+    const uint32_t max_len = remaining_input_length < peek_root_length() ? uint32_t(remaining_input_length) : peek_root_length();
+    if (!raw_json_string_is_quote_terminated(json, max_len)) {
+      return STRING_ERROR;
+    }
+  }
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
   advance_scalar("string");
   return raw_json_string(json+1);
 }
@@ -73970,6 +92210,43 @@ simdjson_warn_unused simdjson_inline simdjson_result<double> value_iterator::get
   return result;
 }

+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_root_float(bool check_trailing) noexcept {
+  auto max_len = peek_root_length();
+  auto json = peek_root_scalar("float");
+  // We use the same buffer size as get_root_double: the number of significant
+  // digits that matter is smaller for binary32, but the JSON document may still
+  // spell out a long number that we must parse (and round) faithfully.
+  uint8_t tmpbuf[1074+8+1+1]; // +1 for null termination.
+  tmpbuf[1074+8+1] = '\0'; // make sure that buffer is always null terminated.
+  if (!_json_iter->copy_to_buffer(json, max_len, tmpbuf, 1074+8+1)) {
+    logger::log_error(*_json_iter, start_position(), depth(), "Root number more than 1082 characters");
+    return NUMBER_ERROR;
+  }
+  auto result = numberparsing::parse_float(tmpbuf);
+  if(result.error() == SUCCESS) {
+    if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+    advance_root_scalar("float");
+  }
+  return result;
+}
+
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_root_float_in_string(bool check_trailing) noexcept {
+  auto max_len = peek_root_length();
+  auto json = peek_root_scalar("float");
+  uint8_t tmpbuf[1074+8+1+1]; // +1 for null termination.
+  tmpbuf[1074+8+1] = '\0'; // make sure that buffer is always null terminated.
+  if (!_json_iter->copy_to_buffer(json, max_len, tmpbuf, 1074+8+1)) {
+    logger::log_error(*_json_iter, start_position(), depth(), "Root number more than 1082 characters");
+    return NUMBER_ERROR;
+  }
+  auto result = numberparsing::parse_float_in_string(tmpbuf);
+  if(result.error() == SUCCESS) {
+    if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+    advance_root_scalar("float");
+  }
+  return result;
+}
+
 simdjson_warn_unused simdjson_inline simdjson_result<bool> value_iterator::get_root_bool(bool check_trailing) noexcept {
   auto max_len = peek_root_length();
   auto json = peek_root_scalar("bool");
@@ -74198,6 +92475,21 @@ simdjson_inline void value_iterator::move_at_container_start() noexcept {
   _json_iter->token.set_position(_start_position + 1);
 }

+simdjson_inline void value_iterator::reenter_at(token_position position, depth_t depth) noexcept {
+  // Unlike reenter_child(), this does not require the live depth to be
+  // exactly one level shallower than depth, nor does it validate against
+  // the parser's per-depth container-start bookkeeping: neither holds in
+  // general for a caller-supplied snapshot (see object_position). What
+  // must still always hold, regardless of what was captured or how far
+  // the live iterator has since moved, is that position and depth are
+  // themselves sane values -- this is the same bound reenter_child()
+  // itself applies unconditionally.
+  SIMDJSON_ASSUME(position != nullptr);
+  SIMDJSON_ASSUME(depth >= 1 && depth < INT32_MAX);
+  _json_iter->_depth = depth;
+  _json_iter->token.set_position(position);
+}
+
 simdjson_inline simdjson_result<bool> value_iterator::reset_array() noexcept {
   if(error()) { return error(); }
   move_at_container_start();
@@ -75732,7 +94024,7 @@ simdjson_inline internal::value128 full_multiplication(uint64_t value1, uint64_t
 /* end file simdjson/fallback/begin.h */
 /* including simdjson/generic/ondemand/amalgamated.h for fallback: #include "simdjson/generic/ondemand/amalgamated.h" */
 /* begin file simdjson/generic/ondemand/amalgamated.h for fallback */
-#if defined(SIMDJSON_CONDITIONAL_INCLUDE) && !defined(SIMDJSON_GENERIC_BUILDER_DEPENDENCIES_H)
+#if defined(SIMDJSON_CONDITIONAL_INCLUDE) && !defined(SIMDJSON_GENERIC_ONDEMAND_DEPENDENCIES_H)
 #error simdjson/generic/ondemand/dependencies.h must be included before simdjson/generic/ondemand/amalgamated.h!
 #endif

@@ -75781,6 +94073,13 @@ class token_iterator;
 class value;
 class value_iterator;

+#if SIMDJSON_SUPPORTS_RANGES
+class array_range;
+class array_range_iterator;
+class object_range;
+class object_range_iterator;
+#endif // SIMDJSON_SUPPORTS_RANGES
+
 } // namespace ondemand
 } // namespace fallback
 } // namespace simdjson
@@ -75813,6 +94112,9 @@ template <> struct is_builtin_deserializable<fallback::ondemand::object> : std::
 template <> struct is_builtin_deserializable<fallback::ondemand::value> : std::true_type {};
 template <> struct is_builtin_deserializable<fallback::ondemand::raw_json_string> : std::true_type {};
 template <> struct is_builtin_deserializable<std::string_view> : std::true_type {};
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template <> struct is_builtin_deserializable<std::u8string_view> : std::true_type {};
+#endif // SIMDJSON_SUPPORTS_CHAR8_T

 template <typename T>
 concept is_builtin_deserializable_v = is_builtin_deserializable<T>::value;
@@ -75830,6 +94132,10 @@ concept nothrow_custom_deserializable = nothrow_tag_invocable<deserialize_tag, V
 template <typename T, typename ValT = fallback::ondemand::value>
 concept nothrow_deserializable = nothrow_custom_deserializable<T, ValT> || is_builtin_deserializable_v<T>;

+// True iff get<T>() on ValT cannot throw: no custom tag_invoke, or that tag_invoke is noexcept.
+template <typename T, typename ValT = fallback::ondemand::value>
+concept nothrow_gettable = !custom_deserializable<T, ValT> || nothrow_custom_deserializable<T, ValT>;
+
 /// Deserialize Tag
 inline constexpr struct deserialize_tag {
   using array_type = fallback::ondemand::array;
@@ -76044,6 +94350,17 @@ public:
    */
   simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> field_key() noexcept;

+  /**
+   * Get the current field's key together with its raw byte length.
+   *
+   * Like field_key(), but also returns the number of raw key bytes (the distance
+   * from the first key byte to the closing quote). The length is recovered from
+   * the structural index -- the next structural token is the ':' -- by stepping
+   * back over any whitespace to the closing quote, avoiding a forward SIMD scan
+   * for the closing quote. Leaves the iterator positioned exactly as field_key().
+   */
+  simdjson_warn_unused simdjson_inline error_code field_key_with_length(raw_json_string &key, std::size_t &len) noexcept;
+
   /**
    * Pass the : in the field and move to its value.
    */
@@ -76196,6 +94513,8 @@ public:
   simdjson_warn_unused simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_warn_unused simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_warn_unused simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<bool> get_bool() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<bool> is_null() noexcept;
   simdjson_warn_unused simdjson_inline bool is_negative() noexcept;
@@ -76214,6 +94533,8 @@ public:
   simdjson_warn_unused simdjson_inline simdjson_result<int64_t> get_root_int64_in_string(bool check_trailing) noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<double> get_root_double(bool check_trailing) noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<double> get_root_double_in_string(bool check_trailing) noexcept;
+  simdjson_warn_unused simdjson_inline simdjson_result<float> get_root_float(bool check_trailing) noexcept;
+  simdjson_warn_unused simdjson_inline simdjson_result<float> get_root_float_in_string(bool check_trailing) noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<bool> get_root_bool(bool check_trailing) noexcept;
   simdjson_warn_unused simdjson_inline bool is_root_negative() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<bool> is_root_integer(bool check_trailing) noexcept;
@@ -76349,6 +94670,15 @@ protected:

   /** @copydoc error_code json_iterator::position() const noexcept; */
   simdjson_inline token_position position() const noexcept;
+  /**
+   * Move the live iterator directly to the given position and depth, without
+   * validating against the parser's per-depth container-start bookkeeping
+   * (unlike json_iterator::reenter_child()). Used to restore a previously
+   * captured mid-container position (see object::revert_position()): that
+   * bookkeeping only tracks each container's own start, not every position
+   * a caller might later capture and revert to, so it does not apply here.
+   */
+  simdjson_inline void reenter_at(token_position position, depth_t depth) noexcept;
   /** @copydoc error_code json_iterator::end_position() const noexcept; */
   simdjson_inline token_position last_position() const noexcept;
   /** @copydoc error_code json_iterator::end_position() const noexcept; */
@@ -76417,9 +94747,12 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+   * Supported types: object, array, raw_json_string, string_view, u8string_view (C++20), uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool
    *
-   * You may use get_double(), get_bool(), get_uint64(), get_int64(),
+   * You may use get_double(), get_float(), get_bool(), get_uint64(), get_int64(),
+   * get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(),
+   * get_float32(), get_float64(),
    * get_object(), get_array(), get_raw_json_string(), or get_string() instead.
    * When SIMDJSON_SUPPORTS_CONCEPTS is set, custom types are also supported.
    *
@@ -76429,7 +94762,7 @@ public:
   template <typename T>
   simdjson_inline simdjson_result<T> get()
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+    noexcept(nothrow_gettable<T, value>)
 #else
     noexcept
 #endif
@@ -76444,7 +94777,8 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+   * Supported types: object, array, raw_json_string, string_view, u8string_view (C++20), uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool
    * If the macro SIMDJSON_SUPPORTS_CONCEPTS is set, then custom types are also supported.
    *
    * @param out This is set to a value of the given type, parsed from the JSON. If there is an error, this may not be initialized.
@@ -76454,7 +94788,7 @@ public:
   template <typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out)
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+    noexcept(nothrow_gettable<T, value>)
 #else
     noexcept
 #endif
@@ -76482,7 +94816,7 @@ public:
       "And you do not seem to have added support for it. Indeed, we have that "
       "simdjson::custom_deserializable<T> is false and the type T is not a default type "
       "such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-      "int64_t, double, or bool.");
+      "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
     static_cast<void>(out); // to get rid of unused errors
     return UNINITIALIZED;
   }
@@ -76491,7 +94825,8 @@ public:
     // immediately fail.
     static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
       "The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-      "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+      "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+      "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
       " get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
       " You may also add support for custom types, see our documentation.");
     static_cast<void>(out); // to get rid of unused errors
@@ -76569,6 +94904,50 @@ public:
    */
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;

+  /**
+   * Cast this JSON value to a 16-bit unsigned integer.
+   *
+   * Calls get_uint64() and checks that the result fits in a uint16_t.
+   *
+   * @returns A 16-bit unsigned integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint16_t.
+   */
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+
+  /**
+   * Cast this JSON value to a 16-bit signed integer.
+   *
+   * Calls get_int64() and checks that the result fits in an int16_t.
+   *
+   * @returns A 16-bit signed integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int16_t.
+   */
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+
+  /**
+   * Cast this JSON value to an 8-bit unsigned integer.
+   *
+   * Calls get_uint64() and checks that the result fits in a uint8_t.
+   *
+   * @returns An 8-bit unsigned integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint8_t.
+   */
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+
+  /**
+   * Cast this JSON value to an 8-bit signed integer.
+   *
+   * Calls get_int64() and checks that the result fits in an int8_t.
+   *
+   * @returns An 8-bit signed integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int8_t.
+   */
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
+
   /**
    * Cast this JSON value to a double.
    *
@@ -76585,6 +94964,53 @@ public:
    */
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;

+  /**
+   * Cast this JSON value to a float (binary32).
+   *
+   * The value is rounded directly to binary32: it is the float nearest to the
+   * JSON number. Note that this may differ from get_double() followed by a cast
+   * to float, which rounds twice.
+   *
+   * @returns A float.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+
+  /**
+   * Cast this JSON value (inside string) to a float (binary32).
+   *
+   * @returns A float.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  /**
+   * Cast this JSON value to a std::float32_t (C++23).
+   *
+   * Same as get_float(): the value is rounded directly to binary32.
+   *
+   * @returns A std::float32_t.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  /**
+   * Cast this JSON value to a std::float64_t (C++23).
+   *
+   * Same as get_double().
+   *
+   * @returns A std::float64_t.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   */
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
+
   /**
    * Cast this JSON value to a string.
    *
@@ -76612,6 +95038,26 @@ public:
    */
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;

+#if SIMDJSON_SUPPORTS_CHAR8_T
+  /**
+   * Cast this JSON value to a C++20 UTF-8 string.
+   *
+   * The string is guaranteed to be valid UTF-8.
+   *
+   * Equivalent to get<std::u8string_view>(). This is a zero-copy alias of
+   * get_string(): the very same bytes are returned, viewed as char8_t.
+   *
+   * Important: a value should be consumed once. Calling get_u8string() twice on the same
+   * value is an error.
+   *
+   * @param allow_replacement Whether to allow a replacement character for unmatched surrogate pairs.
+   * @returns An UTF-8 string. The string is stored in the parser and will be invalidated the next
+   *          time it parses a document or when it is destroyed.
+   * @returns INCORRECT_TYPE if the JSON value is not a string.
+   */
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
   /**
    * Attempts to fill the provided std::string reference with the parsed value of the current string.
    *
@@ -76699,7 +95145,7 @@ public:
   /**
    * Cast this JSON value to an unsigned integer.
    *
-   * @returns A signed 64-bit integer.
+   * @returns A unsigned 64-bit integer.
    * @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit unsigned integer.
    */
   simdjson_inline operator uint64_t() noexcept(false);
@@ -77164,9 +95610,24 @@ public:
   simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template <typename string_type>
   simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -77174,9 +95635,19 @@ public:
   simdjson_inline simdjson_result<bool> get_bool() noexcept;
   simdjson_inline simdjson_result<bool> is_null() noexcept;

-  template<typename T> simdjson_inline simdjson_result<T> get() noexcept;
+  template<typename T> simdjson_inline simdjson_result<T> get()
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, fallback::ondemand::value>);
+#else
+    noexcept;
+#endif

-  template<typename T> simdjson_inline error_code get(T &out) noexcept;
+  template<typename T> simdjson_inline error_code get(T &out)
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, fallback::ondemand::value>);
+#else
+    noexcept;
+#endif

 #if SIMDJSON_EXCEPTIONS
   template <class T>
@@ -77507,6 +95978,7 @@ protected:
   token_position _position{};

   friend class json_iterator;
+  friend class document_stream;
   friend class value_iterator;
   friend class object;
   template <typename... Args>
@@ -77598,6 +96070,9 @@ protected:
    * value of this attribute.
    */
   bool _streaming{false};
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  bool _allow_incomplete_json{false};
+#endif

 public:
   simdjson_inline json_iterator() noexcept = default;
@@ -77622,6 +96097,10 @@ public:
    * start_root_array() and start_root_object().
    */
   simdjson_inline bool streaming() const noexcept;
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  simdjson_inline bool allow_incomplete_json() const noexcept;
+  simdjson_inline size_t remaining_input_length(const uint8_t *json) const noexcept;
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON

   /**
    * Get the root value iterator
@@ -78501,33 +96980,87 @@ public:
    * @param batch_size The batch size to use. MUST be larger than the largest document. The sweet
    *                   spot is cache-related: small enough to fit in cache, yet big enough to
    *                   parse as many documents as possible in one tight loop.
-   *                   Defaults to 10MB, which has been a reasonable sweet spot in our tests.
-   * @param allow_comma_separated (defaults on false) This allows a mode where the documents are
-   *                   separated by commas instead of whitespace. It comes with a performance
-   *                   penalty because the entire document is indexed at once (and the document must be
-   *                   less than 4 GB), and there is no multithreading. In this mode, the batch_size parameter
-   *                   is effectively ignored, as it is set to at least the document size.
+   *                   Defaults to 1MB, which has been a reasonable sweet spot in our tests.
+   * @param allow_comma_separated @deprecated Use stream_format::comma_delimited instead.
+   *                   When true, maps internally to stream_format::comma_delimited.
+   *                   Defaults to false.
    * @return The stream, or an error. An empty input will yield 0 documents rather than an EMPTY error. Errors:
    *         - MEMALLOC if the parser does not have enough capacity and memory allocation fails
    *         - CAPACITY if the parser does not have enough capacity and batch_size > max_capacity.
    *         - other json errors if parsing fails. You should not rely on these errors to always the same for the
    *           same document: they may vary under runtime dispatch (so they may vary depending on your system and hardware).
    */
-  inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
-  inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
-  inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
-  inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size)
+  inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size)
     the string might be automatically padded with up to SIMDJSON_PADDING whitespace characters */
-  inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
-  inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
+  inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @private An rvalue input is destroyed at the end of the full-expression, while the
+   * returned document_stream only holds a pointer to it: iterating the stream would then
+   * read freed memory. These deleted overloads also catch a std::string_view argument,
+   * which would otherwise convert implicitly to a padded_string temporary. */
+  inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size = DEFAULT_BATCH_SIZE) = delete;// unsafe
+  /** @private @overload iterate_many(const std::string &&s, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size = DEFAULT_BATCH_SIZE) = delete;// unsafe
   /** @private We do not want to allow implicit conversion from C string to std::string. */
   simdjson_result<document_stream> iterate_many(const char *buf, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept = delete;

+#ifndef SIMDJSON_DISABLE_DEPRECATED_API
+  /**
+   * @deprecated Use iterate_many with stream_format::comma_delimited instead.
+   */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @private Deleted for the same reason as iterate_many(const std::string &&s, size_t batch_size). */
+  inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size, bool allow_comma_separated) = delete;// unsafe
+  /** @private @overload iterate_many(const std::string &&s, size_t batch_size, bool allow_comma_separated) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size, bool allow_comma_separated) = delete;// unsafe
+#endif // SIMDJSON_DISABLE_DEPRECATED_API
+
+  /**
+   * Parse a stream of JSON documents with explicit format specification.
+   *
+   * @param buf The concatenated JSON documents.
+   * @param len The length of the buffer.
+   * @param batch_size The batch size to use.
+   * @param format The stream format (whitespace_delimited, json_sequence, or comma_delimited).
+   * @return A stream of documents, or an error.
+   */
+  inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(padded_string_view s, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format)
+   *
+   * The string is padded in place if needed (see simdjson::pad), as with iterate_many(std::string &s, size_t batch_size).
+   */
+  inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size, stream_format format) noexcept;
+  /** @private Deleted for the same reason as iterate_many(const std::string &&s, size_t batch_size). */
+  inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+  /** @private @overload iterate_many(const std::string &&s, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+
   /** The capacity of this parser (the largest document it can process). */
   simdjson_pure simdjson_inline size_t capacity() const noexcept;
   /** The maximum capacity of this parser (the largest document it is allowed to process). */
@@ -78655,6 +97188,7 @@ private:
   size_t _capacity{0};
   size_t _max_capacity;
   size_t _max_depth{DEFAULT_MAX_DEPTH};
+  size_t _document_len{0};
   std::unique_ptr<uint8_t[]> string_buf{};

 #if SIMDJSON_DEVELOPMENT_CHECKS
@@ -78717,8 +97251,19 @@ public:
    * Begin array iteration.
    *
    * Part of the std::iterable interface.
+   *
+   * With SIMDJSON_DEVELOPMENT_CHECKS, the iterator locks this array while it is
+   * alive, so that reentrant access (at(), count_elements(), reset(), ...) is
+   * reported as OUT_OF_ORDER_ITERATION.
    */
-  simdjson_inline simdjson_result<array_iterator> begin() noexcept;
+  simdjson_inline simdjson_result<array_iterator> begin() & noexcept;
+  /**
+   * Begin iteration over a temporary array, e.g., `v.get_array().begin()`.
+   *
+   * The iterator does not depend on the array instance and may outlive it, so
+   * it does not lock it.
+   */
+  simdjson_inline simdjson_result<array_iterator> begin() && noexcept;
   /**
    * Sentinel representing the end of the array.
    *
@@ -78849,7 +97394,7 @@ public:
    */
   template <typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out)
-     noexcept(custom_deserializable<T, array> ? nothrow_custom_deserializable<T, array> : true) {
+     noexcept(nothrow_gettable<T, array>) {
     static_assert(custom_deserializable<T, array>);
     return deserialize(*this, out);
   }
@@ -78861,7 +97406,7 @@ public:
    */
   template <typename T>
   simdjson_inline simdjson_result<T> get()
-    noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+    noexcept(nothrow_gettable<T, array>)
   {
     static_assert(std::is_default_constructible<T>::value, "The specified type is not default constructible.");
     T out{};
@@ -78917,6 +97462,10 @@ protected:
    * iter.is_alive() == false indicates iteration is complete.
    */
   value_iterator iter{};
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  bool locked{false};
+  simdjson_inline void set_locked(bool _locked) noexcept;
+#endif

   friend class value;
   friend class document;
@@ -78938,7 +97487,8 @@ public:
   simdjson_inline simdjson_result(error_code error) noexcept; ///< @private
   simdjson_inline simdjson_result() noexcept = default;

-  simdjson_inline simdjson_result<fallback::ondemand::array_iterator> begin() noexcept;
+  simdjson_inline simdjson_result<fallback::ondemand::array_iterator> begin() & noexcept;
+  simdjson_inline simdjson_result<fallback::ondemand::array_iterator> begin() && noexcept;
   simdjson_inline simdjson_result<fallback::ondemand::array_iterator> end() noexcept;
   inline simdjson_result<size_t> count_elements() & noexcept;
   inline simdjson_result<bool> is_empty() & noexcept;
@@ -78958,7 +97508,7 @@ public:
   // TODO: move this code into object-inl.h

   template<typename T>
-  simdjson_inline simdjson_result<T> get() noexcept {
+  simdjson_inline simdjson_result<T> get() noexcept(nothrow_gettable<T, fallback::ondemand::array>) {
     if (error()) { return error(); }
     if constexpr (std::is_same_v<T, fallback::ondemand::array>) {
       return first;
@@ -78966,7 +97516,7 @@ public:
     return first.get<T>();
   }
   template<typename T>
-  simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept {
+  simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept(nothrow_gettable<T, fallback::ondemand::array>) {
     if (error()) { return error(); }
     if constexpr (std::is_same_v<T, fallback::ondemand::array>) {
       out = first;
@@ -79018,6 +97568,15 @@ public:
   /** Create a new, invalid array iterator. */
   simdjson_inline array_iterator() noexcept = default;

+#if SIMDJSON_DEVELOPMENT_CHECKS
+  simdjson_inline ~array_iterator() noexcept;
+
+  simdjson_inline array_iterator(array_iterator&&) noexcept;
+  simdjson_inline array_iterator& operator=(array_iterator&&) noexcept;
+  simdjson_inline array_iterator(const array_iterator&) noexcept;
+  simdjson_inline array_iterator& operator=(const array_iterator&) noexcept;
+#endif
+
   //
   // Iterator interface
   //
@@ -79060,6 +97619,9 @@ public:
 private:
 #if SIMDJSON_DEVELOPMENT_CHECKS
    bool has_been_referenced{false};
+   array* parent{nullptr};
+
+   simdjson_inline array_iterator(const value_iterator &_iter, array* _parent) noexcept;
 #endif
   value_iterator iter{};

@@ -79159,14 +97721,14 @@ public:
   /**
    * Cast this JSON value to an unsigned integer.
    *
-   * @returns A signed 64-bit integer.
+   * @returns A unsigned 64-bit integer.
    * @returns INCORRECT_TYPE If the JSON value is not a 64-bit unsigned integer.
    */
   simdjson_inline simdjson_result<uint64_t> get_uint64() noexcept;
   /**
    * Cast this JSON value (inside string) to an unsigned integer.
    *
-   * @returns A signed 64-bit integer.
+   * @returns A unsigned 64-bit integer.
    * @returns INCORRECT_TYPE If the JSON value is not a 64-bit unsigned integer.
    */
   simdjson_inline simdjson_result<uint64_t> get_uint64_in_string() noexcept;
@@ -79204,6 +97766,46 @@ public:
    * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int32_t.
    */
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  /**
+   * Cast this JSON value to a 16-bit unsigned integer.
+   *
+   * Calls get_uint64() and checks that the result fits in a uint16_t.
+   *
+   * @returns A 16-bit unsigned integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint16_t.
+   */
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  /**
+   * Cast this JSON value to a 16-bit signed integer.
+   *
+   * Calls get_int64() and checks that the result fits in an int16_t.
+   *
+   * @returns A 16-bit signed integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int16_t.
+   */
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  /**
+   * Cast this JSON value to an 8-bit unsigned integer.
+   *
+   * Calls get_uint64() and checks that the result fits in a uint8_t.
+   *
+   * @returns An 8-bit unsigned integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint8_t.
+   */
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  /**
+   * Cast this JSON value to an 8-bit signed integer.
+   *
+   * Calls get_int64() and checks that the result fits in an int8_t.
+   *
+   * @returns An 8-bit signed integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int8_t.
+   */
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   /**
    * Cast this JSON value to a double.
    *
@@ -79219,6 +97821,53 @@ public:
    * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
    */
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+
+  /**
+   * Cast this JSON value to a float (binary32).
+   *
+   * The value is rounded directly to binary32: it is the float nearest to the
+   * JSON number. Note that this may differ from get_double() followed by a cast
+   * to float, which rounds twice.
+   *
+   * @returns A float.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+
+  /**
+   * Cast this JSON value (inside string) to a float (binary32).
+   *
+   * @returns A float.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  /**
+   * Cast this JSON value to a std::float32_t (C++23).
+   *
+   * Same as get_float(): the value is rounded directly to binary32.
+   *
+   * @returns A std::float32_t.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  /**
+   * Cast this JSON value to a std::float64_t (C++23).
+   *
+   * Same as get_double().
+   *
+   * @returns A std::float64_t.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   */
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   /**
    * Cast this JSON value to a string.
    *
@@ -79232,6 +97881,24 @@ public:
    * @returns INCORRECT_TYPE if the JSON value is not a string.
    */
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  /**
+   * Cast this JSON value to a C++20 UTF-8 string.
+   *
+   * The string is guaranteed to be valid UTF-8.
+   *
+   * Equivalent to get<std::u8string_view>(). This is a zero-copy alias of
+   * get_string(): the very same bytes are returned, viewed as char8_t.
+   *
+   * Important: Calling get_u8string() twice on the same document is an error.
+   *
+   * @param allow_replacement Whether to allow a replacement character for unmatched surrogate pairs.
+   * @returns An UTF-8 string. The string is stored in the parser and will be invalidated the next
+   *          time it parses a document or when it is destroyed.
+   * @returns INCORRECT_TYPE if the JSON value is not a string.
+   */
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   /**
    * Attempts to fill the provided std::string reference with the parsed value of the current string.
    *
@@ -79302,9 +97969,12 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool
    *
-   * You may use get_double(), get_bool(), get_uint64(), get_int64(),
+   * You may use get_double(), get_float(), get_bool(), get_uint64(), get_int64(),
+   * get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(),
+   * get_float32(), get_float64(),
    * get_object(), get_array(), get_raw_json_string(), or get_string() instead.
    *
    * @returns A value of the given type, parsed from the JSON.
@@ -79313,7 +97983,7 @@ public:
   template <typename T>
   simdjson_inline simdjson_result<T> get() &
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    noexcept(nothrow_gettable<T, document>)
 #else
     noexcept
 #endif
@@ -79336,7 +98006,7 @@ public:
   template<typename T>
   simdjson_inline simdjson_result<T> get() &&
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    noexcept(nothrow_gettable<T, document>)
 #else
     noexcept
 #endif
@@ -79348,7 +98018,8 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool, value
+   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool, value
    *
    * Be mindful that the document instance must remain in scope while you are accessing object, array and value instances.
    *
@@ -79359,7 +98030,7 @@ public:
   template<typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out) &
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    noexcept(nothrow_gettable<T, document>)
 #else
     noexcept
 #endif
@@ -79372,7 +98043,7 @@ public:
         "And you do not seem to have added support for it. Indeed, we have that "
         "simdjson::custom_deserializable<T> is false and the type T is not a default type "
         "such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-        "int64_t, double, or bool.");
+        "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
       static_cast<void>(out); // to get rid of unused errors
       return UNINITIALIZED;
     }
@@ -79381,7 +98052,8 @@ public:
     // immediately fail.
     static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
       "The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-      "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+      "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+      "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
       " get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
       " You may also add support for custom types, see our documentation.");
     static_cast<void>(out); // to get rid of unused errors
@@ -79390,7 +98062,12 @@ public:
   }

   /** @overload template<typename T> error_code get(T &out) & noexcept */
-  template<typename T> simdjson_deprecated simdjson_inline error_code get(T &out) && noexcept;
+  template<typename T> simdjson_deprecated simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, document>);
+#else
+    noexcept;
+#endif

 #if SIMDJSON_EXCEPTIONS
   /**
@@ -79424,24 +98101,24 @@ public:
   /**
    * Cast this JSON value to an unsigned integer.
    *
-   * @returns A signed 64-bit integer.
+   * @returns A unsigned 64-bit integer.
    * @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit unsigned integer.
    */
-  simdjson_inline operator uint64_t() noexcept(false);
+  explicit simdjson_inline operator uint64_t() noexcept(false);
   /**
    * Cast this JSON value to a signed integer.
    *
    * @returns A signed 64-bit integer.
    * @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit integer.
    */
-  simdjson_inline operator int64_t() noexcept(false);
+  explicit simdjson_inline operator int64_t() noexcept(false);
   /**
    * Cast this JSON value to a double.
    *
    * @returns A double.
    * @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a valid floating-point number.
    */
-  simdjson_inline operator double() noexcept(false);
+  explicit simdjson_inline operator double() noexcept(false);
   /**
    * Cast this JSON value to a string.
    *
@@ -79451,7 +98128,7 @@ public:
    *          time it parses a document or when it is destroyed.
    * @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not a string.
    */
-  simdjson_inline operator std::string_view() noexcept(false) simdjson_lifetime_bound;
+  explicit simdjson_inline operator std::string_view() noexcept(false) simdjson_lifetime_bound;
   /**
    * Cast this JSON value to a raw_json_string.
    *
@@ -79460,14 +98137,14 @@ public:
    * @returns A pointer to the raw JSON for the given string.
    * @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not a string.
    */
-  simdjson_inline operator raw_json_string() noexcept(false) simdjson_lifetime_bound;
+  explicit simdjson_inline operator raw_json_string() noexcept(false) simdjson_lifetime_bound;
   /**
    * Cast this JSON value to a bool.
    *
    * @returns A bool value.
    * @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not true or false.
    */
-  simdjson_inline operator bool() noexcept(false);
+  explicit simdjson_inline operator bool() noexcept(false);
   /**
    * Cast this JSON value to a value when the document is an object or an array.
    *
@@ -79962,9 +98639,24 @@ public:
   simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template <typename string_type>
   simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -79976,7 +98668,7 @@ public:
   template <typename T>
   simdjson_inline simdjson_result<T> get() &
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    noexcept(nothrow_gettable<T, document_reference>)
 #else
     noexcept
 #endif
@@ -79989,7 +98681,8 @@ public:
   template<typename T>
   simdjson_inline simdjson_result<T> get() &&
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    // Forwards to document::get<T>(), so the document customization decides.
+    noexcept(nothrow_gettable<T, document>)
 #else
     noexcept
 #endif
@@ -80001,7 +98694,8 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool, value
+   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool, value
    *
    * Be mindful that the document instance must remain in scope while you are accessing object, array and value instances.
    *
@@ -80012,7 +98706,7 @@ public:
   template<typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out) &
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document_reference> : true)
+    noexcept(nothrow_gettable<T, document_reference>)
 #else
     noexcept
 #endif
@@ -80025,7 +98719,7 @@ public:
         "And you do not seem to have added support for it. Indeed, we have that "
         "simdjson::custom_deserializable<T> is false and the type T is not a default type "
         "such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-        "int64_t, double, or bool.");
+        "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
       static_cast<void>(out); // to get rid of unused errors
       return UNINITIALIZED;
     }
@@ -80034,7 +98728,8 @@ public:
     // immediately fail.
     static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
       "The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-      "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+      "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+      "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
       " get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
       " You may also add support for custom types, see our documentation.");
     static_cast<void>(out); // to get rid of unused errors
@@ -80043,7 +98738,12 @@ public:
   }

   /** @overload template<typename T> error_code get(T &out) & noexcept */
-  template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+  template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, document_reference>);
+#else
+    noexcept;
+#endif
   simdjson_inline simdjson_result<std::string_view> raw_json() noexcept;
 #if SIMDJSON_STATIC_REFLECTION
   template<constevalutil::fixed_string... FieldNames, typename T>
@@ -80056,12 +98756,12 @@ public:
   explicit simdjson_inline operator T() noexcept(false);
   simdjson_inline operator array() & noexcept(false);
   simdjson_inline operator object() & noexcept(false);
-  simdjson_inline operator uint64_t() noexcept(false);
-  simdjson_inline operator int64_t() noexcept(false);
-  simdjson_inline operator double() noexcept(false);
-  simdjson_inline operator std::string_view() noexcept(false);
-  simdjson_inline operator raw_json_string() noexcept(false);
-  simdjson_inline operator bool() noexcept(false);
+  explicit simdjson_inline operator uint64_t() noexcept(false);
+  explicit simdjson_inline operator int64_t() noexcept(false);
+  explicit simdjson_inline operator double() noexcept(false);
+  explicit simdjson_inline operator std::string_view() noexcept(false);
+  explicit simdjson_inline operator raw_json_string() noexcept(false);
+  explicit simdjson_inline operator bool() noexcept(false);
   simdjson_inline operator value() noexcept(false);
 #endif
   simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -80123,9 +98823,24 @@ public:
   simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template <typename string_type>
   simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -80134,11 +98849,31 @@ public:
   simdjson_inline simdjson_result<fallback::ondemand::value> get_value() noexcept;
   simdjson_inline simdjson_result<bool> is_null() noexcept;

-  template<typename T> simdjson_inline simdjson_result<T> get() & noexcept;
-  template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() && noexcept;
+  template<typename T> simdjson_inline simdjson_result<T> get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, fallback::ondemand::document>);
+#else
+    noexcept;
+#endif
+  template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, fallback::ondemand::document>);
+#else
+    noexcept;
+#endif

-  template<typename T> simdjson_inline error_code get(T &out) & noexcept;
-  template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+  template<typename T> simdjson_inline error_code get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, fallback::ondemand::document>);
+#else
+    noexcept;
+#endif
+  template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, fallback::ondemand::document>);
+#else
+    noexcept;
+#endif
 #if SIMDJSON_EXCEPTIONS

   using fallback::implementation_simdjson_result_base<fallback::ondemand::document>::operator*;
@@ -80147,12 +98882,12 @@ public:
   explicit simdjson_inline operator T() noexcept(false);
   simdjson_inline operator fallback::ondemand::array() & noexcept(false);
   simdjson_inline operator fallback::ondemand::object() & noexcept(false);
-  simdjson_inline operator uint64_t() noexcept(false);
-  simdjson_inline operator int64_t() noexcept(false);
-  simdjson_inline operator double() noexcept(false);
-  simdjson_inline operator std::string_view() noexcept(false);
-  simdjson_inline operator fallback::ondemand::raw_json_string() noexcept(false);
-  simdjson_inline operator bool() noexcept(false);
+  explicit simdjson_inline operator uint64_t() noexcept(false);
+  explicit simdjson_inline operator int64_t() noexcept(false);
+  explicit simdjson_inline operator double() noexcept(false);
+  explicit simdjson_inline operator std::string_view() noexcept(false);
+  explicit simdjson_inline operator fallback::ondemand::raw_json_string() noexcept(false);
+  explicit simdjson_inline operator bool() noexcept(false);
   simdjson_inline operator fallback::ondemand::value() noexcept(false);
 #endif
   simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -80218,9 +98953,24 @@ public:
   simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template <typename string_type>
   simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -80229,22 +98979,42 @@ public:
   simdjson_inline simdjson_result<fallback::ondemand::value> get_value() noexcept;
   simdjson_inline simdjson_result<bool> is_null() noexcept;

-  template<typename T> simdjson_inline simdjson_result<T> get() & noexcept;
-  template<typename T> simdjson_inline simdjson_result<T> get() && noexcept;
+  template<typename T> simdjson_inline simdjson_result<T> get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, fallback::ondemand::document_reference>);
+#else
+    noexcept;
+#endif
+  template<typename T> simdjson_inline simdjson_result<T> get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, fallback::ondemand::document_reference>);
+#else
+    noexcept;
+#endif

-  template<typename T> simdjson_inline error_code get(T &out) & noexcept;
-  template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+  template<typename T> simdjson_inline error_code get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, fallback::ondemand::document_reference>);
+#else
+    noexcept;
+#endif
+  template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, fallback::ondemand::document_reference>);
+#else
+    noexcept;
+#endif
 #if SIMDJSON_EXCEPTIONS
   template <class T>
   explicit simdjson_inline operator T() noexcept(false);
   simdjson_inline operator fallback::ondemand::array() & noexcept(false);
   simdjson_inline operator fallback::ondemand::object() & noexcept(false);
-  simdjson_inline operator uint64_t() noexcept(false);
-  simdjson_inline operator int64_t() noexcept(false);
-  simdjson_inline operator double() noexcept(false);
-  simdjson_inline operator std::string_view() noexcept(false);
-  simdjson_inline operator fallback::ondemand::raw_json_string() noexcept(false);
-  simdjson_inline operator bool() noexcept(false);
+  explicit simdjson_inline operator uint64_t() noexcept(false);
+  explicit simdjson_inline operator int64_t() noexcept(false);
+  explicit simdjson_inline operator double() noexcept(false);
+  explicit simdjson_inline operator std::string_view() noexcept(false);
+  explicit simdjson_inline operator fallback::ondemand::raw_json_string() noexcept(false);
+  explicit simdjson_inline operator bool() noexcept(false);
   simdjson_inline operator fallback::ondemand::value() noexcept(false);
 #endif
   simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -80412,10 +99182,7 @@ public:
    *   }
    *   size_t truncated = stream.truncated_bytes();
    *
-   * IMPORTANT: this value is only meaningful under the conditions below. It is
-   * computed from stage-1 bookkeeping, and outside these conditions it is not
-   * merely imprecise, it is arbitrary -- it can exceed size_in_bytes() or wrap
-   * around to a huge value. Check it only when both of the following hold:
+   * IMPORTANT: this value is only meaningful under the conditions below.
    *
    *   - you iterated all the way to the end of the stream;
    *   - no document reported an error. Iteration stops at the first failed
@@ -80424,6 +99191,9 @@ public:
    * If you need to know about a truncated tail outside those conditions, track
    * it yourself from the last successful document (see iterator::current_index()
    * and iterator::source()).
+   *
+   * An empty input (zero bytes) or an input made only of white space contains
+   * no document: truncated_bytes() returns zero.
    */
   inline size_t truncated_bytes() const noexcept;

@@ -80483,7 +99253,10 @@ public:
      *
      * The returned string_view instance is simply a map to the (unparsed)
      * source string: it may thus include white-space characters and all manner
-     * of padding.
+     * of padding. It spans the whole current document, whether or not you
+     * have already accessed (part of) the document. Thus
+     * current_index() + source().size() is the offset just past the end of the
+     * current document, which is useful when reading a stream in chunks.
      *
      * This function (source()) is experimental and the usage
      * may change in future versions of simdjson: we find the API somewhat
@@ -80537,13 +99310,16 @@ private:
    * @param buf is the raw byte buffer we need to process
    * @param len is the length of the raw byte buffer in bytes
    * @param batch_size is the size of the windows (must be strictly greater or equal to the largest JSON document)
+   * @param allow_comma_separated whether to allow comma-separated documents
+   * @param format the stream format
    */
   simdjson_inline document_stream(
     ondemand::parser &parser,
     const uint8_t *buf,
     size_t len,
     size_t batch_size,
-    bool allow_comma_separated
+    bool allow_comma_separated,
+    stream_format format = stream_format::whitespace_delimited
   ) noexcept;

   /**
@@ -80577,8 +99353,23 @@ private:
    */
   inline void next() noexcept;

-  /** Move the json_iterator of the document to the location of the next document in the stream. */
+  /**
+   * Move the json_iterator of the document to the location of the next document
+   * in the stream.
+   *
+   * For formats with a document delimiter (`newline_delimited`, `json_sequence`),
+   * when the iterator is still inside the current document (`depth() > 0`), this
+   * may jump to the next delimiter instead of walking remaining structurals. That
+   * jump does not structure-validate the unread remainder.
+   */
   inline void next_document() noexcept;
+  /** Byte that ends a document under `format`, or 0 if there is none. */
+  simdjson_inline uint8_t document_delimiter() const noexcept;
+  /**
+   * Position the iterator at the first structural at or past the next
+   * `delimiter` in the current batch. Returns false if none is found.
+   */
+  simdjson_inline bool skip_to_delimiter(uint8_t delimiter) noexcept;

   /** Get the next document index. */
   inline size_t next_batch_start() const noexcept;
@@ -80592,6 +99383,7 @@ private:
   size_t len;
   size_t batch_size;
   bool allow_comma_separated;
+  stream_format format;
   /**
    * We are going to use just one document instance. The document owns
    * the json_iterator. It implies that we only ever pass a reference
@@ -80618,7 +99410,7 @@ private:
   /** The error returned from the stage 1 thread. */
   error_code stage1_thread_error{UNINITIALIZED};
   /** The thread used to run stage 1 against the next batch in the background. */
-  std::unique_ptr<stage1_worker> worker{new(std::nothrow) stage1_worker()};
+  std::unique_ptr<stage1_worker> worker{};
   /**
    * The parser used to run stage 1 in the background. Will be swapped
    * with the regular parser when finished.
@@ -80693,6 +99485,16 @@ public:
    * call it again nor can you call key().
    */
   simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  /**
+   * Get the key as a C++20 u8string_view. This is a zero-copy alias of
+   * unescaped_key(): the very same bytes are returned, viewed as char8_t.
+   *
+   * This consumes the key: once you have called unescaped_u8key(), you cannot
+   * call it again nor can you call key().
+   */
+  simdjson_inline simdjson_warn_unused simdjson_result<std::u8string_view> unescaped_u8key(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   /**
    * Get the key as a string_view (for higher speed, consider raw_key).
    * We deliberately use a more cumbersome name (unescaped_key) to force users
@@ -80725,6 +99527,16 @@ public:
    * you can safely call it repeatedly. Use unescaped_key() to get the unescaped key.
    */
   simdjson_inline std::string_view escaped_key() const noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  /**
+   * Get the key as a C++20 u8string_view. This is a zero-copy alias of
+   * escaped_key(): the very same bytes are returned, viewed as char8_t.
+   * The string is unprocessed, so it may contain escape characters
+   * (e.g., \uXXXX or \n). It does not count as a consumption of the content:
+   * you can safely call it repeatedly.
+   */
+  simdjson_inline std::u8string_view escaped_u8key() const noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   /**
    * Get the field value.
    */
@@ -80756,11 +99568,17 @@ public:
   simdjson_inline simdjson_result() noexcept = default;

   simdjson_inline simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> unescaped_u8key(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template<typename string_type>
   simdjson_inline error_code unescaped_key(string_type &receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<fallback::ondemand::raw_json_string> key() noexcept;
   simdjson_inline simdjson_result<std::string_view> key_raw_json_token() noexcept;
   simdjson_inline simdjson_result<std::string_view> escaped_key() noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> escaped_u8key() noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   simdjson_inline simdjson_result<fallback::ondemand::value> value() noexcept;
 };

@@ -80768,6 +99586,1398 @@ public:

 #endif // SIMDJSON_GENERIC_ONDEMAND_FIELD_H
 /* end file simdjson/generic/ondemand/field.h for fallback */
+/* including simdjson/generic/ondemand/key_selector.h for fallback: #include "simdjson/generic/ondemand/key_selector.h" */
+/* begin file simdjson/generic/ondemand/key_selector.h for fallback */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+#define SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #include "simdjson/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/common_defs.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/constevalutil.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/raw_json_string.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#include <array>
+#include <string>      // std::string (key_selector::describe)
+#include <string_view>
+#include <cstddef>
+#include <cstdint>
+#include <cstring>     // std::memcpy (portable unaligned window load)
+#include <utility>     // std::index_sequence (window candidate dispatch)
+#include <type_traits> // std::integral_constant (window candidate dispatch)
+
+// AArch64 only: 32-bit ARM (ARMv7) defines __ARM_NEON too, but lacks the
+// A64-only horizontal reductions (vmaxvq_u32) used below. See issue #2885.
+#if defined(__aarch64__)
+  #include <arm_neon.h>
+  #define SIMDJSON_KEY_SELECTOR_HAS_NEON 1
+#else
+  #define SIMDJSON_KEY_SELECTOR_HAS_NEON 0
+#endif
+#if defined(__SSE2__)
+  #include <emmintrin.h>
+  #define SIMDJSON_KEY_SELECTOR_HAS_SSE2 1
+#else
+  #define SIMDJSON_KEY_SELECTOR_HAS_SSE2 0
+#endif
+#if defined(__loongarch_sx)
+  #include <lsxintrin.h>
+  #define SIMDJSON_KEY_SELECTOR_HAS_LSX 1
+#else
+  #define SIMDJSON_KEY_SELECTOR_HAS_LSX 0
+#endif
+
+#if SIMDJSON_SUPPORTS_CONCEPTS
+
+namespace simdjson {
+namespace fallback {
+namespace ondemand {
+
+namespace key_selector_detail {
+
+// Since not constexpr, triggers compile-time error.
+inline void compile_time_error(const char* message) noexcept { (void)message; }
+
+// ============================================================================
+// Compile-time perfect-hash generator.
+// It scales to ~100 keys at compile time by determining association values one (position, character)
+// symbol at a time (gperf-style) instead of an exhaustive offset search, and
+// falls back to a Hash-and-Displace construction for large/awkward key sets.
+//
+// Only flat tables survive to runtime; the lookup is a few additions plus a
+// single SIMD key comparison (see match_raw below).
+// ============================================================================
+
+// Maximum number of character positions the gperf hash may combine.
+static constexpr std::size_t MAX_POSITIONS = 16;
+// Sentinel "position" meaning "the last character of the key".
+static constexpr std::size_t LAST_CHAR = std::size_t(-1);
+// Runtime-encoded sentinels (stored in uint8 tables).
+static constexpr std::uint8_t POS_LAST_CHAR = 0xFF; // positions_[i] == last char
+static constexpr std::uint8_t HD_MODE       = 0xFF; // num_positions == H&D mode
+// Flags stored in positions[2] in H&D mode to select the key-hash variant.
+static constexpr std::size_t HD_HASH_2BYTE_FLAG = 2;
+static constexpr std::size_t HD_HASH_4BYTE_FLAG = 4;
+
+constexpr std::size_t next_power_of_2(std::size_t n) noexcept {
+    if (n == 0) { return 1; }
+    std::size_t p = 1;
+    while (p < n) { p <<= 1; }
+    return p;
+}
+
+// Character at a given position (LAST_CHAR means last character), or 256 if out
+// of bounds.
+constexpr std::size_t char_at(std::string_view key, std::size_t pos) noexcept {
+    if (pos == LAST_CHAR) {
+        if (key.empty()) { return 256; }
+        return static_cast<unsigned char>(key[key.size() - 1]);
+    }
+    if (pos >= key.size()) { return 256; }
+    return static_cast<unsigned char>(key[pos]);
+}
+
+// Count key pairs that a set of positions fails to distinguish. Keys whose
+// lengths differ modulo the table size are separated by the length term in the
+// hash, so they need no position coverage.
+template <std::size_t N>
+consteval std::size_t count_undistinguished_pairs(
+    const std::array<std::string_view, N>& keys,
+    const std::size_t* positions,
+    std::size_t num_positions,
+    std::size_t modulus) {
+    std::size_t count = 0;
+    for (std::size_t i = 0; i < N; ++i) {
+        for (std::size_t j = i + 1; j < N; ++j) {
+            if (keys[i].size() % modulus != keys[j].size() % modulus) { continue; }
+            bool distinguished = false;
+            for (std::size_t p = 0; p < num_positions; ++p) {
+                if (char_at(keys[i], positions[p]) != char_at(keys[j], positions[p])) {
+                    distinguished = true;
+                    break;
+                }
+            }
+            if (!distinguished) { ++count; }
+        }
+    }
+    return count;
+}
+
+template <std::size_t N>
+consteval bool positions_distinguish(
+    const std::array<std::string_view, N>& keys,
+    const std::size_t* positions,
+    std::size_t num_positions,
+    std::size_t modulus) {
+    return count_undistinguished_pairs<N>(keys, positions, num_positions, modulus) == 0;
+}
+
+// Number of distinct (length % modulus, char_at(key, pos)) pairs at a position.
+template <std::size_t N>
+consteval std::size_t discriminating_power(
+    const std::array<std::string_view, N>& keys,
+    std::size_t pos,
+    std::size_t modulus) {
+    struct pair { std::size_t len_mod; std::size_t ch; };
+    std::array<pair, N> pairs{};
+    for (std::size_t i = 0; i < N; ++i) {
+        pairs[i] = {keys[i].size() % modulus, char_at(keys[i], pos)};
+    }
+    std::size_t count = 0;
+    for (std::size_t i = 0; i < N; ++i) {
+        bool dup = false;
+        for (std::size_t j = 0; j < i; ++j) {
+            if (pairs[i].len_mod == pairs[j].len_mod && pairs[i].ch == pairs[j].ch) {
+                dup = true;
+                break;
+            }
+        }
+        if (!dup) { ++count; }
+    }
+    return count;
+}
+
+template <std::size_t N>
+consteval std::size_t max_key_length(const std::array<std::string_view, N>& keys) {
+    std::size_t m = 0;
+    for (std::size_t i = 0; i < N; ++i) {
+        if (keys[i].size() > m) { m = keys[i].size(); }
+    }
+    return m;
+}
+
+// Bounded backtracking DFS for a minimal set of distinguishing positions.
+template <std::size_t N>
+consteval bool backtracking_search(
+    const std::array<std::string_view, N>& keys,
+    const std::size_t* candidates,
+    std::size_t num_candidates,
+    std::size_t* positions,
+    std::size_t& num_positions_out,
+    std::size_t& budget,
+    std::size_t modulus) {
+    constexpr std::size_t MAX_DEPTH = 8;
+    std::size_t breadth = num_candidates < 20 ? num_candidates : 20;
+
+    struct frame { std::size_t depth; std::size_t next_ci; std::size_t parent_count; };
+    std::array<frame, MAX_DEPTH + 1> stack{};
+    std::size_t sp = 0;
+
+    std::size_t initial_count = count_undistinguished_pairs<N>(keys, positions, 0, modulus);
+    if (budget > 0) { --budget; }
+    if (initial_count == 0) { num_positions_out = 0; return true; }
+
+    stack[0] = {0, 0, initial_count};
+
+    while (budget > 0) {
+        if (sp > MAX_DEPTH) {
+            if (sp == 0) { break; }
+            --sp;
+            ++stack[sp].next_ci;
+            continue;
+        }
+        auto& f = stack[sp];
+        if (f.next_ci >= breadth) {
+            if (sp == 0) { break; }
+            --sp;
+            ++stack[sp].next_ci;
+            continue;
+        }
+        positions[sp] = candidates[f.next_ci];
+        --budget;
+        std::size_t new_count = count_undistinguished_pairs<N>(keys, positions, sp + 1, modulus);
+        if (new_count == 0) { num_positions_out = sp + 1; return true; }
+        if (new_count < f.parent_count && sp + 1 < MAX_DEPTH) {
+            stack[sp + 1] = {sp + 1, f.next_ci + 1, new_count};
+            ++sp;
+        } else {
+            ++f.next_ci;
+        }
+    }
+    return false;
+}
+
+// Phase 1: select character positions that distinguish all colliding pairs.
+template <std::size_t N>
+consteval std::size_t select_positions(
+    const std::array<std::string_view, N>& keys,
+    std::array<std::size_t, MAX_POSITIONS>& positions,
+    std::size_t modulus) {
+    if (positions_distinguish<N>(keys, positions.data(), 0, modulus)) { return 0; }
+
+    std::size_t max_len = max_key_length(keys);
+    constexpr std::size_t MAX_CANDIDATES = 256;
+    std::array<std::size_t, MAX_CANDIDATES> candidates{};
+    std::array<std::size_t, MAX_CANDIDATES> powers{};
+    std::size_t num_candidates = 0;
+    for (std::size_t p = 0; p < max_len && num_candidates < MAX_CANDIDATES - 1; ++p) {
+        candidates[num_candidates] = p;
+        powers[num_candidates] = discriminating_power(keys, p, modulus);
+        ++num_candidates;
+    }
+    if (num_candidates < MAX_CANDIDATES) {
+        candidates[num_candidates] = LAST_CHAR;
+        powers[num_candidates] = discriminating_power(keys, LAST_CHAR, modulus);
+        ++num_candidates;
+    }
+    for (std::size_t i = 0; i < num_candidates; ++i) {
+        for (std::size_t j = i + 1; j < num_candidates; ++j) {
+            if (powers[j] > powers[i]) {
+                auto tc = candidates[i]; candidates[i] = candidates[j]; candidates[j] = tc;
+                auto tp = powers[i]; powers[i] = powers[j]; powers[j] = tp;
+            }
+        }
+    }
+
+    positions[0] = candidates[0];
+    if (positions_distinguish<N>(keys, positions.data(), 1, modulus)) { return 1; }
+
+    positions[0] = 0;
+    positions[1] = LAST_CHAR;
+    if (positions_distinguish<N>(keys, positions.data(), 2, modulus)) { return 2; }
+
+    {
+        std::size_t budget = 5000;
+        std::size_t num_found = 0;
+        if (backtracking_search<N>(keys, candidates.data(), num_candidates,
+                                   positions.data(), num_found, budget, modulus)) {
+            return num_found;
+        }
+    }
+
+    std::size_t num_pos = 0;
+    for (std::size_t ci = 0; ci < num_candidates && num_pos < MAX_POSITIONS; ++ci) {
+        bool already = false;
+        for (std::size_t p = 0; p < num_pos; ++p) {
+            if (positions[p] == candidates[ci]) { already = true; break; }
+        }
+        if (already) { continue; }
+        positions[num_pos] = candidates[ci];
+        ++num_pos;
+        if (positions_distinguish<N>(keys, positions.data(), num_pos, modulus)) { return num_pos; }
+    }
+
+    compile_time_error("Failed to find distinguishing positions for perfect hash");
+    return 0;
+}
+
+// Result of PHF computation. A max-sized slot_to_key array lets the same struct
+// type carry any chosen table size.
+template <std::size_t N>
+struct phf_result {
+    // Allow up to 8x the minimum table size. Sparser tables solve faster.
+    static constexpr std::size_t MAX_TABLE_SIZE = next_power_of_2(N) * 8;
+    std::size_t table_size{};
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso_values{};
+    std::size_t num_positions{};
+    std::array<std::size_t, MAX_POSITIONS> positions{};
+    std::array<std::size_t, MAX_TABLE_SIZE> slot_to_key{};
+};
+
+// Partition-based asso_values search (gperf-style). Determines asso_values one
+// (position, character) symbol at a time; never revisits a value. Equivalence
+// classes (keys sharing the same undetermined symbols) keep the search cheap.
+template <std::size_t N, std::size_t M>
+consteval bool try_generate_gperf(
+    const std::array<std::string_view, N>& keys,
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS>& asso_values,
+    std::size_t& num_positions,
+    std::array<std::size_t, MAX_POSITIONS>& positions,
+    std::array<std::size_t, M>& slot_to_key) {
+    num_positions = select_positions<N>(keys, positions, M);
+
+    for (std::size_t p = 0; p < MAX_POSITIONS; ++p) {
+        for (std::size_t c = 0; c < 256; ++c) { asso_values[p][c] = 0; }
+    }
+
+    if (num_positions == 0) {
+        for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+        for (std::size_t i = 0; i < N; ++i) {
+            std::size_t slot = keys[i].size() % M;
+            if (slot_to_key[slot] != N) { return false; }
+            slot_to_key[slot] = i;
+        }
+        return true;
+    }
+
+    std::array<std::array<std::size_t, MAX_POSITIONS>, N> kchars{};
+    for (std::size_t k = 0; k < N; ++k) {
+        for (std::size_t p = 0; p < num_positions; ++p) {
+            kchars[k][p] = char_at(keys[k], positions[p]);
+        }
+    }
+
+    struct sym_t { std::size_t pos; std::size_t ch; std::size_t freq; };
+    constexpr std::size_t MAX_SYMS = MAX_POSITIONS * 256;
+    std::array<sym_t, MAX_SYMS> syms{};
+    std::size_t nsyms = 0;
+    for (std::size_t p = 0; p < num_positions; ++p) {
+        std::array<std::size_t, 256> freq{};
+        for (std::size_t k = 0; k < N; ++k) {
+            std::size_t c = kchars[k][p];
+            if (c < 256) { freq[c]++; }
+        }
+        for (std::size_t c = 0; c < 256; ++c) {
+            if (freq[c] > 0) { syms[nsyms++] = {p, c, freq[c]}; }
+        }
+    }
+    for (std::size_t i = 0; i < nsyms; ++i) {
+        for (std::size_t j = i + 1; j < nsyms; ++j) {
+            if (syms[j].freq > syms[i].freq) {
+                auto tmp = syms[i]; syms[i] = syms[j]; syms[j] = tmp;
+            }
+        }
+    }
+
+    std::array<std::size_t, N> phash{};
+    for (std::size_t k = 0; k < N; ++k) { phash[k] = keys[k].size(); }
+
+    std::array<std::array<uint64_t, 256>, MAX_POSITIONS> salt{};
+    {
+        uint64_t s = 0x9e3779b97f4a7c15ULL;
+        for (std::size_t p = 0; p < num_positions; ++p) {
+            for (std::size_t c = 0; c < 256; ++c) {
+                s = s * 6364136223846793005ULL + 1442695040888963407ULL;
+                salt[p][c] = s;
+            }
+        }
+    }
+    std::array<uint64_t, N> sig{};
+    for (std::size_t k = 0; k < N; ++k) {
+        uint64_t s = 0;
+        for (std::size_t p = 0; p < num_positions; ++p) {
+            std::size_t c = kchars[k][p];
+            if (c < 256) { s ^= salt[p][c]; }
+        }
+        sig[k] = s;
+    }
+    std::array<std::size_t, N> order{};
+    for (std::size_t k = 0; k < N; ++k) { order[k] = k; }
+
+    std::array<std::size_t, M> slot_gen{};
+    std::size_t gen = 0;
+
+    std::size_t search_limit = next_power_of_2(M);
+    if (search_limit < 32) { search_limit = 32; }
+
+    for (std::size_t si = 0; si < nsyms; ++si) {
+        std::size_t sp = syms[si].pos;
+        std::size_t sc = syms[si].ch;
+
+        uint64_t sp_salt = salt[sp][sc];
+        for (std::size_t k = 0; k < N; ++k) {
+            if (kchars[k][sp] == sc) { sig[k] ^= sp_salt; }
+        }
+
+        for (std::size_t i = 1; i < N; ++i) {
+            std::size_t x = order[i];
+            uint64_t xs = sig[x];
+            std::size_t j = i;
+            while (j > 0 && sig[order[j - 1]] > xs) {
+                order[j] = order[j - 1];
+                --j;
+            }
+            order[j] = x;
+        }
+
+        bool found = false;
+        for (std::size_t v = 0; v < search_limit && !found; ++v) {
+            bool collision = false;
+            std::size_t ci = 0;
+            while (ci < N && !collision) {
+                uint64_t class_sig = sig[order[ci]];
+                std::size_t cj = ci;
+                while (cj < N && sig[order[cj]] == class_sig) { ++cj; }
+                if (cj - ci > 1) {
+                    ++gen;
+                    for (std::size_t x = ci; x < cj; ++x) {
+                        std::size_t k = order[x];
+                        std::size_t h = phash[k];
+                        if (kchars[k][sp] == sc) { h += v; }
+                        h %= M;
+                        if (slot_gen[h] == gen) { collision = true; break; }
+                        slot_gen[h] = gen;
+                    }
+                }
+                ci = cj;
+            }
+            if (!collision) {
+                asso_values[sp][sc] = v;
+                for (std::size_t k = 0; k < N; ++k) {
+                    if (kchars[k][sp] == sc) { phash[k] += v; }
+                }
+                found = true;
+            }
+        }
+        if (!found) { return false; }
+    }
+
+    for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+    for (std::size_t i = 0; i < N; ++i) {
+        std::size_t slot = phash[i] % M;
+        if (slot_to_key[slot] != N) { return false; }
+        slot_to_key[slot] = i;
+    }
+    std::size_t filled = 0;
+    for (std::size_t i = 0; i < M; ++i) {
+        if (slot_to_key[i] != N) { ++filled; }
+    }
+    return filled == N;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_compute_phf(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+    static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso{};
+    std::size_t npos{};
+    std::array<std::size_t, MAX_POSITIONS> pos{};
+    std::array<std::size_t, M> s2k{};
+    if (try_generate_gperf<N, M>(keys, asso, npos, pos, s2k)) {
+        result.table_size = M;
+        result.asso_values = asso;
+        result.num_positions = npos;
+        result.positions = pos;
+        for (std::size_t i = 0; i < M; ++i) { result.slot_to_key[i] = s2k[i]; }
+        for (std::size_t i = M; i < phf_result<N>::MAX_TABLE_SIZE; ++i) { result.slot_to_key[i] = N; }
+        return true;
+    }
+    return false;
+}
+
+template <std::size_t N, std::size_t M, std::size_t MaxM>
+consteval bool try_gperf_po2(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+    if (try_compute_phf<N, M>(keys, result)) { return true; }
+    constexpr std::size_t NextM = M * 2;
+    if constexpr (NextM <= MaxM) { return try_gperf_po2<N, NextM, MaxM>(keys, result); }
+    return false;
+}
+
+// --- Hash-and-Displace fallback --------------------------------------------
+
+constexpr std::size_t hd_bucket_hash(std::string_view key) noexcept {
+    std::size_t c0 = key.empty() ? 0 : static_cast<unsigned char>(key[0]);
+    std::size_t c1 = key.empty() ? 0 : static_cast<unsigned char>(key[key.size() - 1]);
+    return (c0 + c1 * 3 + key.size() * 17) & 0xFF;
+}
+constexpr std::size_t hd_safe_char(const char* p, std::size_t len, std::size_t idx) noexcept {
+    std::size_t has = static_cast<std::size_t>(idx < len);
+    std::size_t si = idx & (std::size_t{0} - has);
+    return static_cast<unsigned char>(p[si]) & (std::size_t{0} - has);
+}
+constexpr std::size_t hd_key_hash_2(std::string_view key) noexcept {
+    std::size_t kc = key.size();
+    kc = kc * 31 + static_cast<unsigned char>(key[0]);
+    kc = kc * 31 + hd_safe_char(key.data(), key.size(), 1);
+    return kc;
+}
+constexpr std::size_t hd_key_hash_4(std::string_view key) noexcept {
+    std::size_t kc = key.size();
+    kc = kc * 31 + static_cast<unsigned char>(key[0]);
+    kc = kc * 31 + hd_safe_char(key.data(), key.size(), 1);
+    kc = kc * 31 + hd_safe_char(key.data(), key.size(), 2);
+    kc = kc * 31 + hd_safe_char(key.data(), key.size(), 3);
+    return kc;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_hash_and_displace(
+    const std::array<std::string_view, N>& keys,
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS>& asso_values,
+    std::size_t& num_positions,
+    std::array<std::size_t, MAX_POSITIONS>& positions,
+    std::array<std::size_t, M>& slot_to_key) {
+    num_positions = select_positions<N>(keys, positions, M);
+
+    if (num_positions == 0) {
+        for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+        for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+        for (std::size_t i = 0; i < N; ++i) {
+            std::size_t slot = keys[i].size() % M;
+            if (slot_to_key[slot] != N) { return false; }
+            slot_to_key[slot] = i;
+        }
+        return true;
+    }
+
+    for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+    num_positions = HD_MODE; // sentinel for H&D mode
+    positions[0] = 0;
+    positions[1] = LAST_CHAR;
+
+    std::array<std::size_t, N> key_bucket{};
+    for (std::size_t i = 0; i < N; ++i) { key_bucket[i] = hd_bucket_hash(keys[i]); }
+
+    struct bucket_info { std::size_t ch; std::size_t count; };
+    std::array<bucket_info, N> buckets{};
+    std::size_t num_buckets = 0;
+    for (std::size_t i = 0; i < N; ++i) {
+        std::size_t bk = key_bucket[i];
+        bool found = false;
+        for (std::size_t b = 0; b < num_buckets; ++b) {
+            if (buckets[b].ch == bk) { ++buckets[b].count; found = true; break; }
+        }
+        if (!found) { buckets[num_buckets++] = {bk, 1}; }
+    }
+    for (std::size_t i = 0; i < num_buckets; ++i) {
+        for (std::size_t j = i + 1; j < num_buckets; ++j) {
+            if (buckets[j].count > buckets[i].count) {
+                auto tmp = buckets[i]; buckets[i] = buckets[j]; buckets[j] = tmp;
+            }
+        }
+    }
+
+    auto try_placement = [&](auto key_hash_fn) -> bool {
+        for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+        for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+        for (std::size_t b = 0; b < num_buckets; ++b) {
+            std::size_t ch = buckets[b].ch;
+            std::array<std::size_t, N> bucket_keys{};
+            std::size_t bk_count = 0;
+            for (std::size_t i = 0; i < N; ++i) {
+                if (key_bucket[i] == ch) { bucket_keys[bk_count++] = i; }
+            }
+            bool placed = false;
+            std::size_t max_d = M < 255 ? M : 255;
+            for (std::size_t d = 0; d < max_d; ++d) {
+                bool ok = true;
+                std::array<std::size_t, N> bucket_slots{};
+                for (std::size_t k = 0; k < bk_count; ++k) {
+                    std::size_t slot = (d + key_hash_fn(keys[bucket_keys[k]])) % M;
+                    if (slot_to_key[slot] != N) { ok = false; break; }
+                    for (std::size_t k2 = 0; k2 < k; ++k2) {
+                        if (bucket_slots[k2] == slot) { ok = false; break; }
+                    }
+                    if (!ok) { break; }
+                    bucket_slots[k] = slot;
+                }
+                if (ok) {
+                    asso_values[0][ch] = d;
+                    for (std::size_t k = 0; k < bk_count; ++k) {
+                        slot_to_key[bucket_slots[k]] = bucket_keys[k];
+                    }
+                    placed = true;
+                    break;
+                }
+            }
+            if (!placed) { return false; }
+        }
+        std::size_t filled = 0;
+        for (std::size_t i = 0; i < M; ++i) {
+            if (slot_to_key[i] != N) { ++filled; }
+        }
+        return filled == N;
+    };
+
+    if (try_placement([](std::string_view k) { return hd_key_hash_2(k); })) {
+        positions[2] = HD_HASH_2BYTE_FLAG;
+        return true;
+    }
+    if (try_placement([](std::string_view k) { return hd_key_hash_4(k); })) {
+        positions[2] = HD_HASH_4BYTE_FLAG;
+        return true;
+    }
+    return false;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_compute_phf_hd(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+    static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso{};
+    std::size_t npos{};
+    std::array<std::size_t, MAX_POSITIONS> pos{};
+    std::array<std::size_t, M> s2k{};
+    if (try_hash_and_displace<N, M>(keys, asso, npos, pos, s2k)) {
+        result.table_size = M;
+        result.asso_values = asso;
+        result.num_positions = npos;
+        result.positions = pos;
+        for (std::size_t i = 0; i < M; ++i) { result.slot_to_key[i] = s2k[i]; }
+        for (std::size_t i = M; i < phf_result<N>::MAX_TABLE_SIZE; ++i) { result.slot_to_key[i] = N; }
+        return true;
+    }
+    return false;
+}
+
+template <std::size_t N, std::size_t M>
+consteval phf_result<N> compute_phf_hd_po2(const std::array<std::string_view, N>& keys) {
+    static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+    phf_result<N> result{};
+    if (try_compute_phf_hd<N, M>(keys, result)) { return result; }
+    constexpr std::size_t NextM = M * 2;
+    if constexpr (NextM <= phf_result<N>::MAX_TABLE_SIZE) {
+        return compute_phf_hd_po2<N, NextM>(keys);
+    } else {
+        compile_time_error("Hash-and-Displace: failed to find valid table size");
+        return result;
+    }
+}
+
+// Compute a perfect hash for `keys`: try gperf at power-of-two sizes (capped so
+// the runtime tables stay within uint8 indices), then fall back to H&D.
+template <std::size_t N>
+consteval phf_result<N> compute_phf(const std::array<std::string_view, N>& keys) {
+    constexpr std::size_t StartM = next_power_of_2(N);
+    constexpr std::size_t GPERF_MAX_TABLE =
+        phf_result<N>::MAX_TABLE_SIZE < 256 ? phf_result<N>::MAX_TABLE_SIZE : 256;
+    if constexpr (StartM <= GPERF_MAX_TABLE) {
+        phf_result<N> result{};
+        if (try_gperf_po2<N, StartM, GPERF_MAX_TABLE>(keys, result)) { return result; }
+    }
+    return compute_phf_hd_po2<N, StartM>(keys);
+}
+
+// ============================================================================
+// Runtime tables (flat, uint8) derived from a phf_result.
+// ============================================================================
+
+template <std::size_t N, std::size_t TableSize, std::size_t MaxKeyLen>
+struct phf_data {
+    std::array<std::array<std::uint8_t, 256>, MAX_POSITIONS> asso_values{};
+    std::array<std::uint8_t, MAX_POSITIONS>                  positions{};
+    std::uint8_t                                             num_positions{};
+    std::uint8_t                                             hd_hash_variant{}; // 2 or 4 (H&D only)
+    std::array<std::uint8_t, TableSize>                      slot_to_key{};
+    // slot_key_bytes[s] holds the key stored at slot s, zero-padded to a 16-byte
+    // multiple so the SIMD comparison can read a whole register.
+    std::array<std::array<char, ((MaxKeyLen + 15) / 16) * 16>, TableSize> slot_key_bytes{};
+    std::array<std::uint8_t, TableSize>                      slot_key_len{};
+};
+
+template <std::size_t N>
+constexpr std::size_t compute_max_key_len(const std::array<std::string_view, N>& keys) noexcept {
+    std::size_t m = 0;
+    for (std::size_t i = 0; i < N; ++i) { if (keys[i].size() > m) { m = keys[i].size(); } }
+    return m;
+}
+
+// Validate keys and build the runtime tables from the computed perfect hash.
+template <std::size_t N, std::size_t TableSize, std::size_t MaxKeyLen>
+consteval phf_data<N, TableSize, MaxKeyLen>
+build_phf_data(const std::array<std::string_view, N>& keys, const phf_result<N>& result) {
+    for (std::size_t i = 0; i < N; ++i) {
+        if (keys[i].empty())            { compile_time_error("empty keys are not allowed in key_selector"); }
+        if (keys[i].size() > MaxKeyLen) { compile_time_error("key length exceeds MaxKeyLen"); }
+        for (char c : keys[i]) {
+            if (c == '\\') { compile_time_error("backslash not allowed in key_selector keys"); }
+            if (c == '"')  { compile_time_error("quote not allowed in key_selector keys"); }
+            if (c == '\0') { compile_time_error("null byte not allowed in key_selector keys"); }
+        }
+        for (std::size_t j = i + 1; j < N; ++j) {
+            if (keys[i] == keys[j]) { compile_time_error("duplicate keys in key_selector"); }
+        }
+    }
+
+    phf_data<N, TableSize, MaxKeyLen> out{};
+
+    if (result.num_positions == HD_MODE) {
+        // H&D mode: single displacement table in asso_values[0].
+        for (std::size_t c = 0; c < 256; ++c) {
+            out.asso_values[0][c] = static_cast<std::uint8_t>(result.asso_values[0][c]);
+        }
+        out.num_positions   = static_cast<std::uint8_t>(HD_MODE);
+        out.hd_hash_variant = static_cast<std::uint8_t>(result.positions[2]);
+    } else {
+        for (std::size_t pi = 0; pi < result.num_positions; ++pi) {
+            for (std::size_t c = 0; c < 256; ++c) {
+                out.asso_values[pi][c] = static_cast<std::uint8_t>(result.asso_values[pi][c] % TableSize);
+            }
+        }
+        out.num_positions = static_cast<std::uint8_t>(result.num_positions);
+        for (std::size_t i = 0; i < result.num_positions; ++i) {
+            out.positions[i] = (result.positions[i] == LAST_CHAR)
+                ? POS_LAST_CHAR
+                : static_cast<std::uint8_t>(result.positions[i]);
+        }
+    }
+
+    for (std::size_t s = 0; s < TableSize; ++s) {
+        out.slot_to_key[s] = static_cast<std::uint8_t>(result.slot_to_key[s]);
+    }
+
+    for (std::size_t s = 0; s < TableSize; ++s) {
+        std::size_t ki = result.slot_to_key[s];
+        if (ki < N) {
+            auto k = keys[ki];
+            out.slot_key_len[s] = static_cast<std::uint8_t>(k.size());
+            for (std::size_t c = 0; c < k.size(); ++c) { out.slot_key_bytes[s][c] = k[c]; }
+        } else {
+            out.slot_key_len[s] = 0; // empty slot: no length can match
+        }
+    }
+    return out;
+}
+
+// --- SIMD runtime primitives ------------------------------------------------
+
+// The runtime matchers read whole 16-byte blocks up to offset 63 (four blocks)
+// for keys as long as the 63-character maximum. That read must stay within the
+// buffer's trailing padding, so the padding has to exceed the largest offset we
+// touch.
+static_assert(SIMDJSON_PADDING > 63,
+              "key_selector requires SIMDJSON_PADDING > 63 for its SIMD key reads");
+
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+// True when every byte of `diff` is zero. A 32-bit-lane horizontal max is
+// enough (zero iff every 32-bit word is zero) and is cheaper than a byte-wide
+// reduction. Do not replace this with a floating-point compare against 0.0:
+// flush-to-zero would treat a denormal as zero.
+simdjson_really_inline bool neon_all_bytes_zero(uint8x16_t diff) noexcept {
+    return vmaxvq_u32(vreinterpretq_u32_u8(diff)) == 0;
+}
+#endif
+
+// Scan for the terminating '"' starting at p. Returns its byte offset (= key
+// length). Caller guarantees SIMDJSON_PADDING (== 64) bytes past the JSON buffer,
+// so reading whole 16-byte blocks up to offset 63 is always safe.
+template <std::size_t MaxKeyLen>
+simdjson_really_inline std::size_t scan_key_length(const char* p) noexcept {
+    // The closing quote of the longest key sits at most at offset MaxKeyLen, so we
+    // scan as many 16-byte blocks as it takes to cover offsets 0..MaxKeyLen. The
+    // cap (63) keeps every covered offset within the 64-byte padding guarantee, so
+    // the SIMD and scalar builds agree.
+    static_assert(MaxKeyLen <= 63, "MaxKeyLen must be <= 63 for current SIMD implementations");
+    // [[maybe_unused]]: the scalar fallback (no NEON/SSE2/LSX) does not use it.
+    [[maybe_unused]] constexpr std::size_t num_blocks = MaxKeyLen / 16 + 1;
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+    for (std::size_t b = 0; b < num_blocks; ++b) {
+        uint8x16_t v = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + b * 16);
+        uint8x16_t cmp = vceqq_u8(v, vdupq_n_u8('"'));
+        uint64_t m = vget_lane_u64(
+            vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(cmp), 4)), 0);
+        if (simdjson_likely(m != 0)) { return b * 16 + (std::size_t(__builtin_ctzll(m)) >> 2); }
+    }
+    return MaxKeyLen + 1;
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+    for (std::size_t b = 0; b < num_blocks; ++b) {
+        __m128i v   = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + b * 16));
+        __m128i cmp = _mm_cmpeq_epi8(v, _mm_set1_epi8('"'));
+        unsigned m  = static_cast<unsigned>(_mm_movemask_epi8(cmp));
+        if (simdjson_likely(m != 0)) { return b * 16 + std::size_t(__builtin_ctz(m)); }
+    }
+    return MaxKeyLen + 1;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+    for (std::size_t b = 0; b < num_blocks; ++b) {
+        __m128i v   = __lsx_vld(reinterpret_cast<const void*>(p + b * 16), 0);
+        __m128i cmp = __lsx_vseq_b(v, __lsx_vreplgr2vr_b('"'));
+        // vmskltz_b gathers the per-byte sign bits (set where the byte equals '"')
+        // into the low 16 bits of lane 0, the LSX equivalent of movemask.
+        unsigned m  = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(cmp), 0)) & 0xFFFFu;
+        if (simdjson_likely(m != 0)) { return b * 16 + std::size_t(__builtin_ctz(m)); }
+    }
+    return MaxKeyLen + 1;
+#else
+    for (std::size_t i = 0; i <= MaxKeyLen; ++i)
+        if (p[i] == '"') return i;
+    return MaxKeyLen + 1;
+#endif
+}
+
+// Byte-equal of p[0..len) against stored[0..len). stored is zero-padded past
+// `len`. Input is read over 16, 32, 48, or 64 bytes (padded JSON buffer
+// guaranteed; SIMDJSON_PADDING == 64).
+template <std::size_t MaxKeyLen>
+simdjson_really_inline bool compare_key_bytes(
+    const char* p, const char* stored, std::size_t len) noexcept {
+    // [[maybe_unused]]: only the NEON/SSE2/LSX branches read these; the scalar
+    // fallback build (no SIMD) leaves them unused, which is an error under -Werror.
+    [[maybe_unused]] alignas(16) static constexpr uint8_t idx16[16] =
+        {0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15};
+    if constexpr (MaxKeyLen <= 16) {
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+        uint8x16_t vp = vld1q_u8(reinterpret_cast<const uint8_t*>(p));
+        uint8x16_t vs = vld1q_u8(reinterpret_cast<const uint8_t*>(stored));
+        uint8x16_t mask = vcltq_u8(vld1q_u8(idx16), vdupq_n_u8(static_cast<uint8_t>(len)));
+        uint8x16_t diff = veorq_u8(vandq_u8(vp, mask), vs);
+        return neon_all_bytes_zero(diff);
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+        __m128i vp = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p));
+        __m128i vs = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored));
+        __m128i idx = _mm_load_si128(reinterpret_cast<const __m128i*>(idx16));
+        __m128i mask = _mm_cmplt_epi8(idx, _mm_set1_epi8(static_cast<char>(len)));
+        __m128i eq = _mm_cmpeq_epi8(_mm_and_si128(vp, mask), vs);
+        return _mm_movemask_epi8(eq) == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+        __m128i vp = __lsx_vld(reinterpret_cast<const void*>(p), 0);
+        __m128i vs = __lsx_vld(reinterpret_cast<const void*>(stored), 0);
+        __m128i idx = __lsx_vld(reinterpret_cast<const void*>(idx16), 0);
+        __m128i mask = __lsx_vslt_b(idx, __lsx_vreplgr2vr_b(static_cast<int>(len)));
+        __m128i eq = __lsx_vseq_b(__lsx_vand_v(vp, mask), vs);
+        return (static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq), 0)) & 0xFFFFu) == 0xFFFFu;
+#else
+        for (std::size_t i = 0; i < len; ++i)
+            if (p[i] != stored[i]) return false;
+        return true;
+#endif
+    } else if constexpr (MaxKeyLen <= 32) {
+        [[maybe_unused]] alignas(16) static constexpr uint8_t idx32_hi[16] =
+            {16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31};
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+        uint8x16_t vp_lo = vld1q_u8(reinterpret_cast<const uint8_t*>(p));
+        uint8x16_t vp_hi = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + 16);
+        uint8x16_t vs_lo = vld1q_u8(reinterpret_cast<const uint8_t*>(stored));
+        uint8x16_t vs_hi = vld1q_u8(reinterpret_cast<const uint8_t*>(stored) + 16);
+        uint8x16_t lenv = vdupq_n_u8(static_cast<uint8_t>(len));
+        uint8x16_t m_lo = vcltq_u8(vld1q_u8(idx16),    lenv);
+        uint8x16_t m_hi = vcltq_u8(vld1q_u8(idx32_hi), lenv);
+        uint8x16_t d_lo = veorq_u8(vandq_u8(vp_lo, m_lo), vs_lo);
+        uint8x16_t d_hi = veorq_u8(vandq_u8(vp_hi, m_hi), vs_hi);
+        return neon_all_bytes_zero(vorrq_u8(d_lo, d_hi));
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+        __m128i vp_lo = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p));
+        __m128i vp_hi = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + 16));
+        __m128i vs_lo = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored));
+        __m128i vs_hi = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored + 16));
+        __m128i lenv = _mm_set1_epi8(static_cast<char>(len));
+        __m128i m_lo = _mm_cmplt_epi8(_mm_load_si128(reinterpret_cast<const __m128i*>(idx16)),    lenv);
+        __m128i m_hi = _mm_cmplt_epi8(_mm_load_si128(reinterpret_cast<const __m128i*>(idx32_hi)), lenv);
+        __m128i eq_lo = _mm_cmpeq_epi8(_mm_and_si128(vp_lo, m_lo), vs_lo);
+        __m128i eq_hi = _mm_cmpeq_epi8(_mm_and_si128(vp_hi, m_hi), vs_hi);
+        return (_mm_movemask_epi8(eq_lo) & _mm_movemask_epi8(eq_hi)) == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+        __m128i lenv = __lsx_vreplgr2vr_b(static_cast<int>(len));
+        __m128i vp_lo = __lsx_vld(reinterpret_cast<const void*>(p), 0);
+        __m128i vp_hi = __lsx_vld(reinterpret_cast<const void*>(p + 16), 0);
+        __m128i vs_lo = __lsx_vld(reinterpret_cast<const void*>(stored), 0);
+        __m128i vs_hi = __lsx_vld(reinterpret_cast<const void*>(stored + 16), 0);
+        __m128i m_lo = __lsx_vslt_b(__lsx_vld(reinterpret_cast<const void*>(idx16), 0),    lenv);
+        __m128i m_hi = __lsx_vslt_b(__lsx_vld(reinterpret_cast<const void*>(idx32_hi), 0), lenv);
+        __m128i eq_lo = __lsx_vseq_b(__lsx_vand_v(vp_lo, m_lo), vs_lo);
+        __m128i eq_hi = __lsx_vseq_b(__lsx_vand_v(vp_hi, m_hi), vs_hi);
+        unsigned mlo = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq_lo), 0)) & 0xFFFFu;
+        unsigned mhi = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq_hi), 0)) & 0xFFFFu;
+        return (mlo & mhi) == 0xFFFFu;
+#else
+        for (std::size_t i = 0; i < len; ++i)
+            if (p[i] != stored[i]) return false;
+        return true;
+#endif
+    } else if constexpr (MaxKeyLen <= 64) {
+        // 3 or 4 16-byte blocks (33..48 -> 3, 49..64 -> 4). stored is zero-padded
+        // to exactly num_blocks*16 bytes (KEY_STRIDE), so neither load overruns it.
+        // [[maybe_unused]]: the scalar fallback (no NEON/SSE2/LSX) does not use it.
+        [[maybe_unused]] constexpr std::size_t num_blocks = (MaxKeyLen + 15) / 16;
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+        uint8x16_t base = vld1q_u8(idx16);
+        uint8x16_t lenv = vdupq_n_u8(static_cast<uint8_t>(len));
+        uint8x16_t acc  = vdupq_n_u8(0);
+        for (std::size_t b = 0; b < num_blocks; ++b) {
+            uint8x16_t vp   = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + b * 16);
+            uint8x16_t vs   = vld1q_u8(reinterpret_cast<const uint8_t*>(stored) + b * 16);
+            uint8x16_t idxv = vaddq_u8(base, vdupq_n_u8(static_cast<uint8_t>(b * 16)));
+            uint8x16_t mask = vcltq_u8(idxv, lenv);
+            acc = vorrq_u8(acc, veorq_u8(vandq_u8(vp, mask), vs));
+        }
+        return neon_all_bytes_zero(acc);
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+        __m128i base = _mm_load_si128(reinterpret_cast<const __m128i*>(idx16));
+        __m128i lenv = _mm_set1_epi8(static_cast<char>(len));
+        int eq = 0xFFFF;
+        for (std::size_t b = 0; b < num_blocks; ++b) {
+            __m128i vp   = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + b * 16));
+            __m128i vs   = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored + b * 16));
+            __m128i idxv = _mm_add_epi8(base, _mm_set1_epi8(static_cast<char>(b * 16)));
+            __m128i mask = _mm_cmplt_epi8(idxv, lenv);
+            eq &= _mm_movemask_epi8(_mm_cmpeq_epi8(_mm_and_si128(vp, mask), vs));
+        }
+        return eq == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+        __m128i base = __lsx_vld(reinterpret_cast<const void*>(idx16), 0);
+        __m128i lenv = __lsx_vreplgr2vr_b(static_cast<int>(len));
+        unsigned acc = 0xFFFFu;
+        for (std::size_t b = 0; b < num_blocks; ++b) {
+            __m128i vp   = __lsx_vld(reinterpret_cast<const void*>(p + b * 16), 0);
+            __m128i vs   = __lsx_vld(reinterpret_cast<const void*>(stored + b * 16), 0);
+            __m128i idxv = __lsx_vadd_b(base, __lsx_vreplgr2vr_b(static_cast<int>(b * 16)));
+            __m128i mask = __lsx_vslt_b(idxv, lenv);
+            __m128i eq   = __lsx_vseq_b(__lsx_vand_v(vp, mask), vs);
+            acc &= static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq), 0)) & 0xFFFFu;
+        }
+        return acc == 0xFFFFu;
+#else
+        for (std::size_t i = 0; i < len; ++i)
+            if (p[i] != stored[i]) return false;
+        return true;
+#endif
+    } else {
+        for (std::size_t i = 0; i < len; ++i)
+            if (p[i] != stored[i]) return false;
+        return true;
+    }
+}
+
+// --- Single 8-bit window fast path ------------------------------------------
+//
+// Many small key sets can be told apart by inspecting a *single* 8-bit window of
+// the key bytes -- and that window need not be byte-aligned. Because every JSON
+// key is terminated by a '"', the bytes at and before a key's length are well
+// defined for any key at least that long: byte i is the key character when i is
+// inside the key and the closing quote when i == len. So we read two bytes at a
+// fixed offset, extract 8 consecutive bits at a fixed intra-byte shift, and if
+// that value is distinct for every key, a 256-entry table maps it straight to a
+// candidate key. The match then needs no hash and -- in the length-free overload
+// -- no SIMD length scan: load two bytes, shift, mask, index the table, and
+// confirm the candidate with one comparison.
+//
+// Allowing an *unaligned* window (a shift of 1..7) mixes bits from two adjacent
+// bytes and discriminates key sets that no single aligned byte can. For example,
+// the partial_tweets keys {created_at,id,text,in_reply_to_status_id,user,
+// retweet_count,favorite_count} share a colliding byte at every aligned position
+// 0,1,2, yet the 8 bits starting at bit offset 2 are unique across all seven.
+// Simpler cases fall out as the shift==0 special case: {"id","screen_name"}
+// splits on byte 0, {"jo","joe"} on the quote at byte 2.
+//
+// The window is confined to the first (shortest key length + 1) bytes so the
+// two-byte read never crosses a key's closing quote into uncontrolled value
+// bytes; that final byte is the shortest key's quote.
+template <std::size_t N, std::size_t MaxKeyLen>
+struct window_data {
+    static constexpr std::size_t KEY_STRIDE = ((MaxKeyLen + 15) / 16) * 16;
+    bool                                            ok{false};
+    std::uint8_t                                    byte_offset{0}; // first byte of the 2-byte read
+    std::uint8_t                                    shift{0};       // intra-byte bit shift (0..7)
+    std::array<std::uint8_t, 256>                   window_to_key{}; // window byte -> key index, N if none
+    std::array<std::uint8_t, N>                     key_len{};
+    std::array<std::array<char, KEY_STRIDE>, N>     key_bytes{};     // zero-padded
+};
+
+// Byte seen at `idx` for key `i`: a key character when idx is inside the key, or
+// the closing '"' when idx == len (callers keep idx <= every key's length).
+template <std::size_t N>
+constexpr unsigned window_byte_at(const std::array<std::string_view, N>& keys,
+                                  std::size_t i, std::size_t idx) noexcept {
+    if (idx < keys[i].size()) { return static_cast<unsigned char>(keys[i][idx]); }
+    return static_cast<unsigned>('"');
+}
+
+// The 8-bit window value for key `i` at (byte_offset, shift): the little-endian
+// pair (byte[off], byte[off+1]) shifted right by `shift` and truncated. Matches
+// the runtime read exactly.
+template <std::size_t N>
+constexpr unsigned window_value(const std::array<std::string_view, N>& keys,
+                                std::size_t i, std::size_t byte_offset, std::size_t shift) noexcept {
+    unsigned lo = window_byte_at<N>(keys, i, byte_offset);
+    unsigned hi = window_byte_at<N>(keys, i, byte_offset + 1);
+    return ((lo | (hi << 8)) >> shift) & 0xFFu;
+}
+
+template <std::size_t N, std::size_t MaxKeyLen>
+consteval window_data<N, MaxKeyLen>
+compute_window(const std::array<std::string_view, N>& keys) {
+    window_data<N, MaxKeyLen> out{};
+
+    std::size_t min_len = keys[0].size();
+    for (std::size_t i = 1; i < N; ++i) {
+        if (keys[i].size() < min_len) { min_len = keys[i].size(); }
+    }
+
+    // Iterate windows nearest the front first (cheapest to read, smallest shift).
+    for (std::size_t off = 0; off <= min_len; ++off) {
+        for (std::size_t shift = 0; shift < 8; ++shift) {
+            // The read touches byte off, and byte off+1 when shift != 0. Both must
+            // stay within the safe region [0, min_len] (min_len is the shortest
+            // key's quote index). off <= min_len is guaranteed by the loop bound.
+            if (shift != 0 && off + 1 > min_len) { continue; }
+
+            bool distinct = true;
+            for (std::size_t i = 0; i < N && distinct; ++i) {
+                for (std::size_t j = i + 1; j < N; ++j) {
+                    if (window_value<N>(keys, i, off, shift) == window_value<N>(keys, j, off, shift)) {
+                        distinct = false;
+                        break;
+                    }
+                }
+            }
+            if (!distinct) { continue; }
+
+            out.ok          = true;
+            out.byte_offset = static_cast<std::uint8_t>(off);
+            out.shift       = static_cast<std::uint8_t>(shift);
+            for (std::size_t b = 0; b < 256; ++b) { out.window_to_key[b] = static_cast<std::uint8_t>(N); }
+            for (std::size_t i = 0; i < N; ++i) {
+                out.window_to_key[window_value<N>(keys, i, off, shift)] = static_cast<std::uint8_t>(i);
+                out.key_len[i] = static_cast<std::uint8_t>(keys[i].size());
+                for (std::size_t c = 0; c < keys[i].size(); ++c) { out.key_bytes[i][c] = keys[i][c]; }
+            }
+            return out;
+        }
+    }
+    return out; // ok == false: no single 8-bit window distinguishes the keys
+}
+
+// Read the 8-bit window at (byte_offset, shift) from a padded key pointer.
+simdjson_really_inline std::uint8_t read_window(const char* p, std::size_t byte_offset,
+                                                std::size_t shift) noexcept {
+    std::uint16_t w;
+    // Two controlled bytes (within the shortest key + its quote, hence within the
+    // padded buffer). memcpy is the portable little-endian unaligned load.
+    std::memcpy(&w, p + byte_offset, sizeof(w));
+#if defined(__BYTE_ORDER__) && (__BYTE_ORDER__ == __ORDER_BIG_ENDIAN__)
+    w = static_cast<std::uint16_t>((w >> 8) | (w << 8));
+#endif
+    return static_cast<std::uint8_t>((w >> shift) & 0xFFu);
+}
+
+// Verify a window candidate. The window table already mapped the key to the
+// single possible index `ki` (< N); here we confirm it. Folding over 0..N-1
+// turns the runtime `ki` into a compile-time index in the matching arm, so the
+// candidate's length and bytes are constants for compare_key_bytes -- the same
+// specialization the ordered find_field path gets from a CT-length
+// unsafe_is_equal, and what keeps this path competitive. The closing-quote check
+// (p[L] == '"') both confirms the key ends exactly at the candidate's length and
+// rejects a wrong-length key, so this works whether or not the caller knew len.
+template <std::size_t N, std::size_t MaxKeyLen, std::size_t... Is>
+simdjson_really_inline std::size_t
+match_window_candidate(const char* p, std::uint8_t ki,
+                       const window_data<N, MaxKeyLen>& w,
+                       std::index_sequence<Is...>) noexcept {
+  std::size_t result = N;
+  auto try_match = [&](auto Ic) {
+    constexpr std::size_t i = decltype(Ic)::value;
+    if (ki == i && p[w.key_len[i]] == '"' &&
+        key_selector_detail::compare_key_bytes<MaxKeyLen>(
+            p, w.key_bytes[i].data(), w.key_len[i])) {
+      result = i;
+    }
+  };
+  (try_match(std::integral_constant<std::size_t, Is>{}), ...);
+  return result;
+}
+
+// --- describe() string helpers (constexpr; no std::to_string, which is not) ---
+
+// Append the decimal form of v to s.
+SIMDJSON_CONSTEXPR_STRING void append_uint(std::string& s, std::size_t v) {
+    if (v == 0) { s.push_back('0'); return; }
+    char buf[20];
+    std::size_t n = 0;
+    while (v > 0) { buf[n++] = static_cast<char>('0' + (v % 10)); v /= 10; }
+    while (n > 0) { s.push_back(buf[--n]); }
+}
+
+// Append a byte as its decimal value, plus the printable character in quotes
+// when it is in the printable ASCII range (e.g. "110 ('n')").
+SIMDJSON_CONSTEXPR_STRING void append_byte(std::string& s, unsigned b) {
+    append_uint(s, b);
+    if (b >= 0x20 && b < 0x7f) {
+        s += " ('";
+        s.push_back(static_cast<char>(b));
+        s += "')";
+    }
+}
+
+} // namespace key_selector_detail
+
+/**
+ * Stateless, compile-time key selector.
+ *
+ * Usage:
+ *   using sel_t = key_selector<"id", "text", "user">;
+ *   std::size_t i = sel_t::match_raw(raw_key); // returns sel_t::size() on miss
+ *
+ * The perfect hash is built at compile time (gperf-style, with a
+ * Hash-and-Displace fallback) and only flat tables survive to runtime. All
+ * tables are static constexpr, so the lookup fully inlines.
+ *
+ * Limitations:
+ *   - Each key must be at most 63 characters long (and no longer than
+ *     SIMDJSON_PADDING). Longer keys trigger a compile-time error.
+ *   - The number of keys should be moderate. The hard limit is 255 keys;
+ *     compilation time grows with the number of keys, so prefer a few dozen at
+ *     most per selector.
+ *   - Keys must be distinct, non-empty, and free of backslash, double-quote and
+ *     null bytes (matching is done against the raw, unescaped JSON key bytes).
+ */
+template <constevalutil::fixed_string... Keys>
+struct key_selector {
+    static constexpr std::size_t N = sizeof...(Keys);
+    static_assert(N > 0,   "key_selector requires at least one key");
+    static_assert(N <= 255,"key_selector supports at most 255 keys");
+
+    static constexpr std::array<std::string_view, N> keys{ Keys.view()... };
+    static constexpr std::size_t max_key_len = key_selector_detail::compute_max_key_len<N>(keys);
+    static_assert(max_key_len <= SIMDJSON_PADDING,
+                  "key longer than SIMDJSON_PADDING is not supported");
+    // The SIMD key-length scan covers offsets 0..63 (four 16-byte blocks), which
+    // stays within the 64-byte padding guarantee. A 64-character key's closing
+    // quote would land at offset 64 and be missed on NEON/SSE2/LSX while still
+    // matching in scalar builds, so cap at 63 to keep implementations in agreement.
+    static_assert(max_key_len <= 63,
+                  "key_selector keys must be at most 63 characters long");
+
+    static constexpr auto result = key_selector_detail::compute_phf<N>(keys);
+    static constexpr std::size_t table_size = result.table_size;
+
+    static constexpr auto phf =
+        key_selector_detail::build_phf_data<N, table_size, max_key_len>(keys, result);
+
+    // Single 8-bit-window discriminator (when one exists). Detected at compile
+    // time and selected with `if constexpr` below, so the hash path is compiled
+    // out for key sets that qualify, and this is compiled out for those that do
+    // not.
+    static constexpr auto window =
+        key_selector_detail::compute_window<N, max_key_len>(keys);
+
+    static constexpr std::size_t size() noexcept { return N; }
+
+    /**
+     * Look up a JSON key whose length is already known. p must point at the first
+     * key byte (just after the opening quote) in a padded simdjson buffer, and len
+     * must be the number of raw key bytes (the distance to the closing quote).
+     * Returns the selector index in [0, N) on match, or N on miss.
+     *
+     * Prefer this overload when the caller can obtain the key length cheaply (for
+     * example, object::for_each derives it from the structural index rather than
+     * re-scanning for the closing quote).
+     */
+    static simdjson_really_inline std::size_t match_raw(const char* p, std::size_t len) noexcept {
+        if (len == 0 || len > max_key_len) { return N; }
+
+        if constexpr (window.ok) {
+            // One 8-bit window selects the only possible candidate key;
+            // match_window_candidate confirms it (bytes + closing quote). p sits
+            // in a padded buffer and the window stays within the shortest key +
+            // quote, so the two-byte read is always in bounds. len is unused here
+            // because the quote check already pins the key's end.
+            std::uint8_t ki = window.window_to_key[
+                key_selector_detail::read_window(p, window.byte_offset, window.shift)];
+            if (ki >= N) { return N; }
+            return key_selector_detail::match_window_candidate(
+                p, ki, window, std::make_index_sequence<N>{});
+        }
+
+        std::size_t slot;
+        if (phf.num_positions == key_selector_detail::HD_MODE) {
+            // Hash-and-Displace: bucket displacement + per-key hash.
+            std::string_view key(p, len);
+            std::size_t bucket = key_selector_detail::hd_bucket_hash(key);
+            std::size_t kh = (phf.hd_hash_variant == key_selector_detail::HD_HASH_2BYTE_FLAG)
+                ? key_selector_detail::hd_key_hash_2(key)
+                : key_selector_detail::hd_key_hash_4(key);
+            slot = (phf.asso_values[0][bucket] + kh) & (table_size - 1);
+        } else {
+            // gperf: h = len + sum of asso_values over the selected positions.
+            // positions / num_positions / asso_values are compile-time constants,
+            // so this loop fully unrolls. The idx < len guard mirrors the
+            // generator's char_at()-> 256 -> skip behavior for out-of-range
+            // positions (required: arbitrary positions may exceed a key's length).
+            std::size_t h = len;
+            for (std::uint8_t i = 0; i < phf.num_positions; ++i) {
+                std::uint8_t pos = phf.positions[i];
+                std::size_t idx = (pos == key_selector_detail::POS_LAST_CHAR)
+                                  ? (len - std::size_t{1})
+                                  : static_cast<std::size_t>(pos);
+                if (idx < len) {
+                    h += phf.asso_values[i][static_cast<unsigned char>(p[idx])];
+                }
+            }
+            slot = h & (table_size - 1);
+        }
+
+        std::uint8_t ki = phf.slot_to_key[slot];
+        if (ki >= N) { return N; }
+        if (phf.slot_key_len[slot] != len) { return N; }
+        if (!key_selector_detail::compare_key_bytes<max_key_len>(
+                p, phf.slot_key_bytes[slot].data(), len)) { return N; }
+        return ki;
+    }
+
+    /**
+     * Look up a JSON key. rjs must point just after an opening quote in a padded
+     * simdjson buffer. Returns the selector index in [0, N) on match, or N on miss.
+     * The key length is recovered with a SIMD scan for the closing quote; callers
+     * that already know the length should use the (p, len) overload above.
+     */
+    static simdjson_really_inline std::size_t match_raw(raw_json_string rjs) noexcept {
+        const char* p = rjs.raw();
+        if constexpr (window.ok) {
+            // One 8-bit window picks the candidate; verifying the candidate's
+            // bytes and its closing '"' confirms the full key, so the length scan
+            // is unnecessary. The window read is in bounds (padding), and the
+            // candidate length is at most max_key_len.
+            std::uint8_t ki = window.window_to_key[
+                key_selector_detail::read_window(p, window.byte_offset, window.shift)];
+            if (ki >= N) { return N; }
+            return key_selector_detail::match_window_candidate(
+                p, ki, window, std::make_index_sequence<N>{});
+        }
+        return match_raw(p, key_selector_detail::scan_key_length<max_key_len>(p));
+    }
+
+    /** Return the key text at selector index i (i in [0, N)). */
+    static constexpr std::string_view key_at(std::size_t i) noexcept {
+        return keys[i];
+    }
+
+    /**
+     * Return a complete, human-readable, multi-line description of how this
+     * selector classifies a key: which algorithm was selected at compile time
+     * (single 8-bit window, gperf-style perfect hash, or hash-and-displace), the
+     * exact bytes/positions it inspects, and the contents of the lookup tables
+     * (which window bytes or hash slots map to which key). The text mirrors what
+     * match_raw() does step by step.
+     *
+     * Everything it reports is derived from the compile-time tables, so describe()
+     * is itself usable in a constant expression when the standard library supports
+     * constexpr std::string (__cpp_lib_constexpr_string):
+     *
+     *   static_assert(!key_selector<"name", "city">::describe().empty());
+     *
+     * It allocates a std::string and is meant for documentation, debugging and
+     * tests, not for any hot path.
+     */
+    static SIMDJSON_CONSTEXPR_STRING std::string describe() {
+        std::string s;
+        s += "key_selector: ";
+        key_selector_detail::append_uint(s, N);
+        s += " keys, max key length ";
+        key_selector_detail::append_uint(s, max_key_len);
+        s += "\nkeys:\n";
+        for (std::size_t i = 0; i < N; ++i) {
+            s += "  [";
+            key_selector_detail::append_uint(s, i);
+            s += "] \"";
+            s += keys[i];
+            s += "\" (length ";
+            key_selector_detail::append_uint(s, keys[i].size());
+            s += ")\n";
+        }
+        if constexpr (window.ok) {
+            // Mirrors the window fast path of match_raw().
+            s += "algorithm: single 8-bit window\n";
+            s += "  step 1: read 2 bytes at offset ";
+            key_selector_detail::append_uint(s, window.byte_offset);
+            s += ", interpret them as a little-endian 16-bit value, shift right by ";
+            key_selector_detail::append_uint(s, window.shift);
+            s += " bits, and keep the low 8 bits\n";
+            s += "  step 2: map that byte through a 256-entry table to a key index (";
+            key_selector_detail::append_uint(s, N);
+            s += " means no match):\n";
+            for (std::size_t b = 0; b < 256; ++b) {
+                if (window.window_to_key[b] < N) {
+                    s += "    byte ";
+                    key_selector_detail::append_byte(s, static_cast<unsigned>(b));
+                    s += " -> key ";
+                    key_selector_detail::append_uint(s, window.window_to_key[b]);
+                    s += "\n";
+                }
+            }
+            s += "  step 3: confirm the candidate by checking the closing quote sits at the key's length and comparing the key bytes\n";
+        } else {
+            // Mirrors the perfect-hash path of match_raw().
+            if constexpr (phf.num_positions == key_selector_detail::HD_MODE) {
+                s += "algorithm: hash-and-displace perfect hash\n";
+                s += "  step 1: bucket = (first_byte + last_byte*3 + length*17) mod 256\n";
+                s += "  step 2: keyhash = base-31 rolling hash of the length and the first ";
+                key_selector_detail::append_uint(s, phf.hd_hash_variant);
+                s += " bytes\n";
+                s += "  step 3: slot = (displacement[bucket] + keyhash) mod ";
+                key_selector_detail::append_uint(s, table_size);
+                s += "\n  step 4: slot_to_key[slot] gives the candidate key index\n";
+                s += "  per-key derivation:\n";
+                for (std::size_t i = 0; i < N; ++i) {
+                    std::string_view k = keys[i];
+                    std::size_t bucket = key_selector_detail::hd_bucket_hash(k);
+                    std::size_t kh = (phf.hd_hash_variant == key_selector_detail::HD_HASH_2BYTE_FLAG)
+                        ? key_selector_detail::hd_key_hash_2(k)
+                        : key_selector_detail::hd_key_hash_4(k);
+                    std::size_t slot = (phf.asso_values[0][bucket] + kh) & (table_size - 1);
+                    s += "    \"";
+                    s += k;
+                    s += "\": bucket=";
+                    key_selector_detail::append_uint(s, bucket);
+                    s += " displacement=";
+                    key_selector_detail::append_uint(s, phf.asso_values[0][bucket]);
+                    s += " keyhash=";
+                    key_selector_detail::append_uint(s, kh);
+                    s += " slot=";
+                    key_selector_detail::append_uint(s, slot);
+                    s += "\n";
+                }
+            } else {
+                s += "algorithm: gperf-style perfect hash over ";
+                key_selector_detail::append_uint(s, phf.num_positions);
+                s += " character position(s)\n";
+                s += "  step 1: h = key length\n";
+                s += "  step 2: for each position below, add its association value for the key byte there (a position past the key's length contributes 0):\n";
+                for (std::size_t i = 0; i < phf.num_positions; ++i) {
+                    s += "    position ";
+                    if (phf.positions[i] == key_selector_detail::POS_LAST_CHAR) {
+                        s += "last character";
+                    } else {
+                        s += "byte index ";
+                        key_selector_detail::append_uint(s, phf.positions[i]);
+                    }
+                    s += "\n";
+                }
+                s += "  step 3: slot = h mod ";
+                key_selector_detail::append_uint(s, table_size);
+                s += " (a power of two, applied as a bitmask)\n";
+                s += "  step 4: slot_to_key[slot] gives the candidate key index\n";
+                s += "  per-key derivation:\n";
+                for (std::size_t i = 0; i < N; ++i) {
+                    std::string_view k = keys[i];
+                    std::size_t h = k.size();
+                    for (std::size_t pi = 0; pi < phf.num_positions; ++pi) {
+                        std::size_t pos = phf.positions[pi];
+                        std::size_t idx = (pos == key_selector_detail::POS_LAST_CHAR)
+                                          ? (k.size() - 1) : pos;
+                        if (idx < k.size()) {
+                            h += phf.asso_values[pi][static_cast<unsigned char>(k[idx])];
+                        }
+                    }
+                    std::size_t slot = h & (table_size - 1);
+                    s += "    \"";
+                    s += k;
+                    s += "\": h=";
+                    key_selector_detail::append_uint(s, h);
+                    s += " slot=";
+                    key_selector_detail::append_uint(s, slot);
+                    s += "\n";
+                }
+            }
+            s += "  occupied slots (slot -> key):\n";
+            for (std::size_t slot = 0; slot < table_size; ++slot) {
+                if (phf.slot_to_key[slot] < N) {
+                    s += "    slot ";
+                    key_selector_detail::append_uint(s, slot);
+                    s += " -> key ";
+                    key_selector_detail::append_uint(s, phf.slot_to_key[slot]);
+                    s += " (\"";
+                    s += keys[phf.slot_to_key[slot]];
+                    s += "\", length ";
+                    key_selector_detail::append_uint(s, phf.slot_key_len[slot]);
+                    s += ")\n";
+                }
+            }
+            s += "  confirm the candidate by checking the key length matches and comparing the key bytes\n";
+        }
+        return s;
+    }
+};
+
+namespace key_selector_detail {
+template <typename> struct is_key_selector : std::false_type {};
+template <constevalutil::fixed_string... Keys>
+struct is_key_selector<key_selector<Keys...>> : std::true_type {};
+} // namespace key_selector_detail
+
+/**
+ * Matches any instantiation of key_selector<Keys...>. Used to constrain
+ * object::for_each so that passing a non-selector type yields a clear
+ * constraint error rather than a cascade of failures inside for_each.
+ */
+template <typename T>
+concept key_selector_type = key_selector_detail::is_key_selector<T>::value;
+
+} // namespace ondemand
+} // namespace fallback
+} // namespace simdjson
+
+#endif // SIMDJSON_SUPPORTS_CONCEPTS
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+/* end file simdjson/generic/ondemand/key_selector.h for fallback */
 /* including simdjson/generic/ondemand/object.h for fallback: #include "simdjson/generic/ondemand/object.h" */
 /* begin file simdjson/generic/ondemand/object.h for fallback */
 #ifndef SIMDJSON_GENERIC_ONDEMAND_OBJECT_H
@@ -80777,6 +100987,7 @@ public:
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/implementation_simdjson_result_base.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/key_selector.h" */
 /* amalgamation skipped (editor-only): #include <vector> */
 /* amalgamation skipped (editor-only): #if SIMDJSON_STATIC_REFLECTION && SIMDJSON_SUPPORTS_CONCEPTS */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_string_builder.h"  // for constevalutil::fixed_string */
@@ -80787,6 +100998,114 @@ namespace simdjson {
 namespace fallback {
 namespace ondemand {

+#if SIMDJSON_SUPPORTS_CONCEPTS
+/**
+ * Result of object::for_each: the first error encountered (SUCCESS if none) and
+ * the number of distinct selector keys that matched during the walk. A
+ * matched_count equal to Selector::size() means every selected key was present
+ * in the object. Implicitly converts to error_code so existing callers that only
+ * care about the error (including SIMDJSON_TRY and the test ASSERT_* macros) keep
+ * working unchanged.
+ *
+ * On error paths (a handler returns a non-SUCCESS error_code, or a direct-target
+ * deserialization via value::get fails), matched_count reflects only the number
+ * of distinct selected keys that were successfully handled *before* the failure;
+ * the failing key is not counted. This is the current behavior.
+ *
+ * Marked [[nodiscard]]: for_each surfaces parse/type errors (from a matched
+ * value, a returning handler, or the structural walk itself) only through this
+ * result, so it must not be silently dropped. This matters most in builds
+ * without exceptions, where it is the only channel for those errors. To
+ * deliberately ignore it, assign to a variable or cast to void.
+ */
+struct [[nodiscard]] for_each_result {
+  error_code error{SUCCESS};
+  std::size_t matched_count{0};
+  constexpr operator error_code() const noexcept { return error; }
+};
+
+namespace key_selector_for_each_detail {
+/**
+ * A per-key handler for the variadic object::for_each is either:
+ *   - an invocable taking a value (run custom logic for that field), or
+ *   - a deserialization target T, in which case the matched value is assigned
+ *     directly via value::get(T&).
+ * The target form lets callers bind fields straight to variables without a
+ * lambda per key -- e.g. obj.for_each<"name","city","age">(name, city, age) --
+ * while still allowing a lambda in any position when custom logic is needed
+ * (for example to descend into a nested object).
+ */
+template <typename H>
+concept field_handler =
+    std::is_invocable_v<std::remove_reference_t<H>&, value> ||
+    ::simdjson::deserializable<std::remove_cvref_t<H>, value>;
+
+/**
+ * noexcept-ness of handling a single handler: nothrow-invocability for the
+ * callback form, or nothrow-deserializability for the direct-target form
+ * (builtin scalar/string targets are noexcept; custom targets follow their
+ * own tag_invoke noexcept specification).
+ */
+template <typename H>
+inline constexpr bool nothrow_field_handler_v =
+    std::is_invocable_v<std::remove_reference_t<H>&, value>
+        ? std::is_nothrow_invocable_v<std::remove_reference_t<H>&, value>
+        : ::simdjson::nothrow_deserializable<std::remove_cvref_t<H>, value>;
+
+/**
+ * nothrow_field_handler_v folded over a whole handler pack. The variadic
+ * for_each overloads spell their conditional-noexcept specifier with this
+ * single id-expression rather than an inline fold expression. MSVC (VS18)
+ * mishandles a fold-expression that appears directly in the noexcept-specifier
+ * of an out-of-line template definition: it fails to recognize the definition's
+ * specifier as matching the in-class declaration's and reports a spurious
+ * C2382 ("redefinition; different exception specifications"). Naming the fold
+ * here keeps the specifier a plain identifier, which matches reliably.
+ */
+template <typename... Handlers>
+inline constexpr bool nothrow_field_handlers_v =
+    (nothrow_field_handler_v<Handlers> && ...);
+} // namespace key_selector_for_each_detail
+#endif
+
+/**
+ * An opaque snapshot of an object's scanning position, obtained from
+ * object::get_current_position() and consumed by object::revert_position().
+ *
+ * This bundles the raw token position with the iteration depth that was
+ * live at the moment of capture: a scalar field value (e.g. a number)
+ * unwinds back to the object's own depth once fully consumed, while a
+ * compound value (array/object) not yet fully consumed stays one level
+ * deeper. The correct depth to restore to therefore depends on what was
+ * live when the snapshot was taken, not on the object itself, which is
+ * why both pieces travel together here rather than being recomputed later.
+ *
+ * The members are private and only constructible by object itself: a
+ * hand-built value here would let revert_position() jump to an arbitrary,
+ * unvalidated position and depth, and this type carries no way to check
+ * that a given snapshot still refers to a live object. Only ever pass
+ * along a value you got from get_current_position(), and only back to the
+ * same object, before that object has been reset() or the parser has
+ * iterate()d a new document: neither invalidates a snapshot in a way this
+ * type can detect.
+ */
+struct object_position {
+  /**
+   * Default-constructed so a variable can be declared and assigned later,
+   * matching e.g. document()/object(). Not a valid position to revert to.
+   */
+  simdjson_inline object_position() noexcept = default;
+
+private:
+  token_position position{};
+  depth_t depth{};
+
+  simdjson_inline object_position(token_position position_, depth_t depth_) noexcept
+    : position(position_), depth(depth_) {}
+
+  friend class object;
+};
+
 /**
  * A forward-only JSON object field iterator.
  */
@@ -80805,8 +101124,19 @@ public:
    * Using the iterator directly is also possible but error-prone and discouraged. In particular,
    * you must dereference the iterator exactly once per iteration (before calling '++').
    * Doing otherwise is unsafe and may lead to errors. You are responsible for ensuring
+   *
+   * With SIMDJSON_DEVELOPMENT_CHECKS, the iterator locks this object while it is
+   * alive, so that reentrant access (find_field(), reset(), ...) is reported as
+   * OUT_OF_ORDER_ITERATION.
+   */
+  simdjson_inline simdjson_result<object_iterator> begin() & noexcept;
+  /**
+   * Get an iterator to the start of a temporary object, e.g., `v.get_object().begin()`.
+   *
+   * The iterator does not depend on the object instance and may outlive it, so
+   * it does not lock it.
    */
-  simdjson_inline simdjson_result<object_iterator> begin() noexcept;
+  simdjson_inline simdjson_result<object_iterator> begin() && noexcept;
   simdjson_inline simdjson_result<object_iterator> end() noexcept;
   /**
    * Look up a field by name on an object (order-sensitive). By order-sensitive, we mean that
@@ -80818,10 +101148,11 @@ public:
    *
    * ```cpp
    * simdjson::ondemand::parser parser;
-   * auto obj = parser.parse(R"( { "x": 1, "y": 2, "z": 3 } )"_padded);
-   * double z = obj.find_field("z");
-   * double y = obj.find_field("y");
-   * double x = obj.find_field("x");
+   * auto json = R"( { "x": 1, "y": 2, "z": 3 } )"_padded;
+   * auto doc = parser.iterate(json);
+   * double z = doc.find_field("z");
+   * double y = doc.find_field("y");
+   * double x = doc.find_field("x");
    * ```
    * If you have multiple fields with a matching key ({"x": 1,  "x": 1}) be mindful
    * that only one field is returned.
@@ -80894,6 +101225,100 @@ public:
   /** @overload simdjson_inline simdjson_result<value> find_field_unordered(std::string_view key) & noexcept; */
   simdjson_inline simdjson_result<value> operator[](std::string_view key) && noexcept;

+#if SIMDJSON_SUPPORTS_CONCEPTS
+  /**
+   * Walk this object once and invoke on_match(selector_index, value) for each
+   * field whose key is in the compile-time key_selector Selector, in JSON order
+   * (first occurrence of a duplicate key wins). Iteration stops once all
+   * Selector::size() keys have matched or the object ends. The value is consumed
+   * in place, so this is a low-overhead way to extract a known set of fields
+   * regardless of their order in the JSON.
+   *
+   * Like other object iteration in simdjson, for_each consumes the object by
+   * advancing the underlying iterator state; after the call the same object
+   * instance should not be used for further field access or iteration.
+   *
+   * Usage:
+   *   using sel_t = ondemand::key_selector<"id", "text", "user">;
+   *   obj.for_each<sel_t>([&](std::size_t i, ondemand::value v) {
+   *     switch (i) { case 0: ...; case 1: ...; }
+   *   });
+   *
+   * Limitations (see key_selector): each key must be at most 63 characters long,
+   * and the number of keys should be moderate (hard limit 255; a handful is
+   * best, as the compile-time perfect hash may fail or slow compilation for
+   * large key sets). The keys must be distinct, non-empty, and free of backslash, double-quote and
+   * null bytes.
+   *
+   * The callback may return either void or an error_code. When it returns an
+   * error_code, the walk stops at the first non-SUCCESS result and that error is
+   * returned, which lets the callback surface value-parse errors.
+   *
+   * This function is conditionally noexcept: it is noexcept exactly when invoking
+   * the callback is noexcept. The callback runs inside this frame, so a throwing
+   * callback (e.g. one using the exception-throwing conversions like
+   * std::string_view(value) or uint64_t(value)) makes for_each potentially
+   * throwing too -- the exception propagates to the caller instead of crossing a
+   * noexcept boundary and calling std::terminate.
+   *
+   * @returns a for_each_result holding the first error encountered while walking
+   *          the object (including any error returned by the callback, SUCCESS if
+   *          none) and the number of distinct selector keys that matched. The
+   *          result converts implicitly to error_code, so callers that only need
+   *          the error can ignore the count.
+   */
+  template <typename Selector, typename Func>
+    requires key_selector_type<Selector> &&
+             std::is_invocable_v<Func&, std::size_t, value>
+  simdjson_inline for_each_result for_each(Func&& on_match)
+      noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, value>);
+
+  /**
+   * Variadic per-key form. Provide exactly one handler per key in the Selector
+   * (compiler-enforced). Handlers are processed in JSON document order for the
+   * matching keys. Each handler is either:
+   *   - a deserialization target (a variable), in which case the matched value
+   *     is assigned to it via value::get -- no lambda required; or
+   *   - an invocable taking the ondemand::value (for custom logic such as
+   *     descending into a nested object). It may return void or error_code;
+   *     returning error_code lets you surface parse/type errors.
+   * The two styles may be mixed freely, one handler per key.
+   *
+   * Example (bind fields straight to variables):
+   *   using fields = ondemand::key_selector<"name", "city", "age">;
+   *   obj.for_each<fields>(name, city, age);
+   *
+   * Example (mixing a target and a lambda):
+   *   obj.for_each<ondemand::key_selector<"id", "user">>(
+   *     id,                                          // assigned via value::get
+   *     [&](ondemand::value v){ u = read_user(v); }  // custom logic
+   *   );
+   *
+   * The index-based single-callback form (taking (size_t, value)) remains
+   * available for shared-state or more complex per-key logic.
+   */
+  template <typename Selector, typename... Handlers>
+    requires key_selector_type<Selector> &&
+             (sizeof...(Handlers) == Selector::size()) &&
+             (key_selector_for_each_detail::field_handler<Handlers> && ...)
+  simdjson_inline for_each_result for_each(Handlers&&... on_match)
+      noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
+  /**
+   * Direct-key shorthand. Equivalent to for_each<key_selector<Keys...>>(...).
+   * Lets you write the keys inline without a separate using/alias, binding each
+   * field straight to a variable (or a lambda, see the Selector form above):
+   *
+   *   obj.for_each<"name", "city", "age">(name, city, age);
+   */
+  template <constevalutil::fixed_string... Keys, typename... Handlers>
+    requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+             (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+             (key_selector_for_each_detail::field_handler<Handlers> && ...)
+  simdjson_inline for_each_result for_each(Handlers&&... on_match)
+      noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+#endif
+
   /**
    * Get the value associated with the given JSON pointer. We use the RFC 6901
    * https://tools.ietf.org/html/rfc6901 standard, interpreting the current node
@@ -80970,6 +101395,34 @@ public:
    * @returns true if the object contains some elements (not empty)
    */
   inline simdjson_result<bool> reset() & noexcept;
+  /**
+   * Get an opaque token representing the object's current scanning position.
+   * Pass it to revert_position() to return to this exact point later, without
+   * paying the cost of a full reset() and re-scan from the beginning.
+   *
+   * A typical use is an optional field that may or may not be next: capture
+   * the position, attempt find_field(), and on NO_SUCH_FIELD, revert_position()
+   * instead of reset() so that fields already consumed are not rescanned.
+   *
+   * The returned token is only valid for this object, and only until it is
+   * reset() or the parser iterate()s a new document; using it after either
+   * is undefined behavior (see object_position).
+   *
+   * @returns An opaque position token.
+   */
+  simdjson_inline object_position get_current_position() const noexcept;
+  /**
+   * Return the object's scanning position to a snapshot previously obtained
+   * from get_current_position(). Unlike reset(), this does not rescan the
+   * object from the beginning: fields before the captured position remain
+   * consumed, and scanning resumes exactly where the snapshot was captured.
+   *
+   * @param position A snapshot previously returned by get_current_position(),
+   *        for this same object.
+   * @returns SUCCESS, or OUT_OF_ORDER_ITERATION if called during active
+   *          iteration (SIMDJSON_DEVELOPMENT_CHECKS builds only).
+   */
+  simdjson_inline error_code revert_position(object_position position) noexcept;
   /**
    * This method scans the beginning of the object and checks whether the
    * object is empty.
@@ -81015,7 +101468,7 @@ public:
    */
   template <typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out)
-     noexcept(custom_deserializable<T, object> ? nothrow_custom_deserializable<T, object> : true) {
+     noexcept(nothrow_gettable<T, object>) {
     static_assert(custom_deserializable<T, object>);
     return deserialize(*this, out);
   }
@@ -81027,7 +101480,7 @@ public:
    */
   template <typename T>
   simdjson_inline simdjson_result<T> get()
-    noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+    noexcept(nothrow_gettable<T, object>)
   {
     static_assert(std::is_default_constructible<T>::value, "The specified type is not default constructible.");
     T out{};
@@ -81079,10 +101532,18 @@ protected:
   simdjson_warn_unused simdjson_inline error_code find_field_raw(const std::string_view key) noexcept;

   value_iterator iter{};
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  bool locked{false};
+  simdjson_inline void set_locked(bool _locked) noexcept;
+#endif

   friend class value;
   friend class document;
   friend struct simdjson_result<object>;
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  friend class object_iterator;
+  friend struct simdjson_result<object_iterator>;
+#endif
 };

 } // namespace ondemand
@@ -81098,7 +101559,8 @@ public:
   simdjson_inline simdjson_result(error_code error) noexcept; ///< @private
   simdjson_inline simdjson_result() noexcept = default;

-  simdjson_inline simdjson_result<fallback::ondemand::object_iterator> begin() noexcept;
+  simdjson_inline simdjson_result<fallback::ondemand::object_iterator> begin() & noexcept;
+  simdjson_inline simdjson_result<fallback::ondemand::object_iterator> begin() && noexcept;
   simdjson_inline simdjson_result<fallback::ondemand::object_iterator> end() noexcept;
   simdjson_inline simdjson_result<fallback::ondemand::value> find_field(std::string_view key) & noexcept;
   simdjson_inline simdjson_result<fallback::ondemand::value> find_field(std::string_view key) && noexcept;
@@ -81116,6 +101578,8 @@ public:
 #endif
   simdjson_inline error_code for_each_at_path_with_wildcard(std::string_view json_path, Func&& callback) noexcept;
   inline simdjson_result<bool> reset() noexcept;
+  inline simdjson_result<fallback::ondemand::object_position> get_current_position() noexcept;
+  inline error_code revert_position(fallback::ondemand::object_position position) noexcept;
   inline simdjson_result<bool> is_empty() noexcept;
   inline simdjson_result<size_t> count_fields() & noexcept;
   inline simdjson_result<std::string_view> raw_json() noexcept;
@@ -81123,7 +101587,7 @@ public:
   // TODO: move this code into object-inl.h

   template<typename T>
-  simdjson_inline simdjson_result<T> get() noexcept {
+  simdjson_inline simdjson_result<T> get() noexcept(nothrow_gettable<T, fallback::ondemand::object>) {
     if (error()) { return error(); }
     if constexpr (std::is_same_v<T, fallback::ondemand::object>) {
       return first;
@@ -81131,7 +101595,7 @@ public:
     return first.get<T>();
   }
   template<typename T>
-  simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept {
+  simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept(nothrow_gettable<T, fallback::ondemand::object>) {
     if (error()) { return error(); }
     if constexpr (std::is_same_v<T, fallback::ondemand::object>) {
       out = first;
@@ -81141,6 +101605,39 @@ public:
     return SUCCESS;
   }

+  /**
+   * Forwards to object::for_each on the underlying object, so error-code-style
+   * chains (e.g. doc["x"].get_object()) can call for_each without first
+   * extracting the object. If this result holds an error, that error is returned
+   * (with a zero match count) and the callback is not invoked. See
+   * object::for_each for the semantics.
+   */
+  template <typename Selector, typename Func>
+    requires fallback::ondemand::key_selector_type<Selector> &&
+             std::is_invocable_v<Func&, std::size_t, fallback::ondemand::value>
+  simdjson_inline fallback::ondemand::for_each_result for_each(Func&& on_match)
+      noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, fallback::ondemand::value>);
+
+  /**
+   * Forwarding overload for the variadic per-key form (targets and/or lambdas).
+   */
+  template <typename Selector, typename... Handlers>
+    requires fallback::ondemand::key_selector_type<Selector> &&
+             (sizeof...(Handlers) == Selector::size()) &&
+             (fallback::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+  simdjson_inline fallback::ondemand::for_each_result for_each(Handlers&&... on_match)
+      noexcept(fallback::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
+  /**
+   * Forwarding overload for the direct-key variadic form.
+   */
+  template <constevalutil::fixed_string... Keys, typename... Handlers>
+    requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+             (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+             (fallback::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+  simdjson_inline fallback::ondemand::for_each_result for_each(Handlers&&... on_match)
+      noexcept(fallback::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
 #if SIMDJSON_STATIC_REFLECTION
   // TODO: move this code into object-inl.h
   template<constevalutil::fixed_string... FieldNames, typename T>
@@ -81181,6 +101678,15 @@ public:
    */
   simdjson_inline object_iterator() noexcept = default;

+#if SIMDJSON_DEVELOPMENT_CHECKS
+   simdjson_inline ~object_iterator() noexcept;
+
+   simdjson_inline object_iterator(object_iterator&&) noexcept;
+   simdjson_inline object_iterator& operator=(object_iterator&&) noexcept;
+   simdjson_inline object_iterator(const object_iterator&) noexcept;
+   simdjson_inline object_iterator& operator=(const object_iterator&) noexcept;
+#endif
+
   //
   // Iterator interface
   //
@@ -81200,6 +101706,9 @@ public:
 private:
 #if SIMDJSON_DEVELOPMENT_CHECKS
    bool has_been_referenced{false};
+   object* parent{nullptr};
+
+   simdjson_inline object_iterator(const value_iterator &_iter, object* _parent) noexcept;
 #endif
   /**
    * The underlying JSON iterator.
@@ -81245,6 +101754,191 @@ public:

 #endif // SIMDJSON_GENERIC_ONDEMAND_OBJECT_ITERATOR_H
 /* end file simdjson/generic/ondemand/object_iterator.h for fallback */
+/* including simdjson/generic/ondemand/ranges.h for fallback: #include "simdjson/generic/ondemand/ranges.h" */
+/* begin file simdjson/generic/ondemand/ranges.h for fallback */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_RANGES_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_ONDEMAND_RANGES_H */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/field.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#if SIMDJSON_SUPPORTS_RANGES
+
+namespace simdjson {
+namespace fallback {
+namespace ondemand {
+
+/**
+ * A ranges-compatible iterator adapter for JSON arrays.
+ *
+ * Wraps array_iterator to satisfy std::input_iterator by providing:
+ * - const operator* (via mutable internal state)
+ * - post-increment operator
+ * - iterator_concept tag
+ *
+ * The mutable approach is standard for single-pass input iterators that
+ * read from external sources (similar to std::istream_iterator).
+ */
+class array_range_iterator {
+public:
+  using iterator_concept = std::input_iterator_tag;
+  using value_type = simdjson_result<value>;
+  using reference = simdjson_result<value>;
+  using difference_type = std::ptrdiff_t;
+
+  simdjson_inline array_range_iterator() noexcept = default;
+  simdjson_inline explicit array_range_iterator(array_iterator iter) noexcept;
+
+  /**
+   * Get the current element. Const-qualified for std::indirectly_readable;
+   * internally delegates to the mutable wrapped iterator.
+   */
+  simdjson_inline simdjson_result<value> operator*() const noexcept;
+  simdjson_inline array_range_iterator& operator++() noexcept;
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+  simdjson_inline void operator++(int) noexcept;
+SIMDJSON_POP_DISABLE_WARNINGS
+  /**
+   * Comparison delegates to array_iterator::operator==, which checks
+   * whether the underlying parser has finished the array (depth-based).
+   */
+  simdjson_inline friend bool operator==(const array_range_iterator& a,
+                                         const array_range_iterator& b) noexcept {
+    return a.iter_ == b.iter_;
+  }
+
+private:
+  mutable array_iterator iter_{};
+};
+
+/**
+ * A std::ranges::view over a JSON array.
+ *
+ * Wraps an ondemand::array and exposes begin()/end() that return
+ * array_range_iterator (satisfying std::input_iterator), enabling
+ * use with std::views::transform and other range adaptors.
+ *
+ * If the array's begin() returns an error (only possible under
+ * SIMDJSON_DEVELOPMENT_CHECKS), the range will be empty and error()
+ * will return the error code.
+ *
+ * Usage:
+ *   ondemand::parser parser;
+ *   auto doc = parser.iterate(json);
+ *   auto arr = doc.get_array().value();
+ *   for (auto elem : ondemand::get_range(arr)) { ... }
+ */
+class array_range {
+public:
+  simdjson_inline array_range() noexcept = default;
+  simdjson_inline explicit array_range(array& arr) noexcept;
+
+  simdjson_inline array_range_iterator begin() noexcept;
+  simdjson_inline array_range_iterator end() noexcept;
+
+  /** Returns SUCCESS if the range was created successfully, or the error code otherwise. */
+  simdjson_inline error_code error() const noexcept { return error_; }
+
+private:
+  array_iterator begin_{};
+  array_iterator end_{};
+  error_code error_{SUCCESS};
+};
+
+/**
+ * A ranges-compatible iterator adapter for JSON objects.
+ *
+ * Wraps object_iterator to satisfy std::input_iterator, yielding
+ * simdjson_result<field> elements (key-value pairs).
+ */
+class object_range_iterator {
+public:
+  using iterator_concept = std::input_iterator_tag;
+  using value_type = simdjson_result<field>;
+  using reference = simdjson_result<field>;
+  using difference_type = std::ptrdiff_t;
+
+  simdjson_inline object_range_iterator() noexcept = default;
+  simdjson_inline explicit object_range_iterator(object_iterator iter) noexcept;
+
+  simdjson_inline simdjson_result<field> operator*() const noexcept;
+  simdjson_inline object_range_iterator& operator++() noexcept;
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+  simdjson_inline void operator++(int) noexcept;
+SIMDJSON_POP_DISABLE_WARNINGS
+  simdjson_inline friend bool operator==(const object_range_iterator& a,
+                                         const object_range_iterator& b) noexcept {
+    return a.iter_ == b.iter_;
+  }
+
+private:
+  mutable object_iterator iter_{};
+};
+
+/**
+ * A std::ranges::view over a JSON object.
+ *
+ * Wraps an ondemand::object and exposes begin()/end() that return
+ * object_range_iterator, enabling use with range adaptors.
+ *
+ * If the object's begin() returns an error, the range will be empty
+ * and error() will return the error code.
+ */
+class object_range {
+public:
+  simdjson_inline object_range() noexcept = default;
+  simdjson_inline explicit object_range(object& obj) noexcept;
+
+  simdjson_inline object_range_iterator begin() noexcept;
+  simdjson_inline object_range_iterator end() noexcept;
+
+  /** Returns SUCCESS if the range was created successfully, or the error code otherwise. */
+  simdjson_inline error_code error() const noexcept { return error_; }
+
+private:
+  object_iterator begin_{};
+  object_iterator end_{};
+  error_code error_{SUCCESS};
+};
+
+/** Get a std::ranges compatible view over a JSON array. */
+simdjson_inline array_range get_range(array& arr) noexcept;
+
+/** Get a std::ranges compatible view over a JSON object (key-value pairs). */
+simdjson_inline object_range get_key_value_range(object& obj) noexcept;
+
+#if SIMDJSON_EXCEPTIONS
+/** Get a std::ranges compatible view, unwrapping the simdjson_result (throws on error). */
+simdjson_inline array_range get_range(simdjson_result<array> result);
+
+/** Get a std::ranges compatible view, unwrapping the simdjson_result (throws on error). */
+simdjson_inline object_range get_key_value_range(simdjson_result<object> result);
+#endif // SIMDJSON_EXCEPTIONS
+
+} // namespace ondemand
+} // namespace fallback
+} // namespace simdjson
+
+namespace std {
+namespace ranges {
+template<>
+inline constexpr bool enable_view<simdjson::fallback::ondemand::array_range> = true;
+template<>
+inline constexpr bool enable_view<simdjson::fallback::ondemand::object_range> = true;
+} // namespace ranges
+} // namespace std
+
+#endif // SIMDJSON_SUPPORTS_RANGES
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_RANGES_H
+/* end file simdjson/generic/ondemand/ranges.h for fallback */
 /* including simdjson/generic/ondemand/serialization.h for fallback: #include "simdjson/generic/ondemand/serialization.h" */
 /* begin file simdjson/generic/ondemand/serialization.h for fallback */
 #ifndef SIMDJSON_GENERIC_ONDEMAND_SERIALIZATION_H
@@ -81377,12 +102071,14 @@ inline std::ostream& operator<<(std::ostream& out, simdjson::simdjson_result<sim
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/annotations.h" */
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */

 #include <concepts>
 #include <limits>
 #if SIMDJSON_STATIC_REFLECTION
 #include <meta>
+#include <vector>
 // #include <static_reflection> // for std::define_static_string - header not available yet
 #endif

@@ -81407,10 +102103,25 @@ error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept {

 template <std::floating_point T>
 error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept {
-  double x;
-  SIMDJSON_TRY(val.get_double().get(x));
-  out = static_cast<T>(x);
-  return SUCCESS;
+  if constexpr (std::is_same_v<T, float>) {
+    // Going through binary64 and then rounding to binary32 would round twice
+    // and could produce a value that is not the float nearest to the JSON
+    // number, so we parse to binary32 directly.
+    return val.get_float().get(out);
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  } else if constexpr (std::is_same_v<T, std::float32_t>) {
+    // Same reason as float.
+    float x;
+    SIMDJSON_TRY(val.get_float().get(x));
+    out = static_cast<T>(x);
+    return SUCCESS;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+  } else {
+    double x;
+    SIMDJSON_TRY(val.get_double().get(x));
+    out = static_cast<T>(x);
+    return SUCCESS;
+  }
 }

 template <std::signed_integral T>
@@ -81446,11 +102157,59 @@ template <concepts::constructible_from_string_view T, typename ValT>
 error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(std::is_nothrow_constructible_v<T, std::string_view>) {
   std::string_view str;
   SIMDJSON_TRY(val.get_string().get(str));
-  out = T{str};
+  if constexpr (requires { out.assign(str.data(), str.size()); }) {
+    // Copy straight into out (e.g., std::string): building a temporary and
+    // move-assigning it is markedly slower.
+    out.assign(str.data(), str.size());
+  } else {
+    out = T{str};
+  }
+  return SUCCESS;
+}
+
+#if SIMDJSON_SUPPORTS_CHAR8_T
+// any C++20 char8_t string-like type (can be constructed from std::u8string_view),
+// such as std::u8string
+template <concepts::constructible_from_u8string_view T, typename ValT>
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(std::is_nothrow_constructible_v<T, std::u8string_view>) {
+  std::u8string_view str;
+  SIMDJSON_TRY(val.get_u8string().get(str));
+  if constexpr (requires { out.assign(str.data(), str.size()); }) {
+    // Copy straight into out (e.g., std::u8string), as for std::string above.
+    out.assign(str.data(), str.size());
+  } else {
+    out = T{str};
+  }
   return SUCCESS;
 }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T


+namespace details {
+// Whether to deserialize the elements of the container T directly into a new
+// element (emplace_one(out) and then get) rather than into a temporary that is
+// then moved into the container. A temporary of a trivially copyable type lives
+// in registers and is cheap to move, and value-initializing a small element in
+// place costs more than it saves (GCC zeroes it with rep stos). But moving a
+// large element, or a string (whose deserialization then needs a temporary
+// string and a move assignment of its own), is expensive.
+template <typename T>
+concept deserialize_in_place =
+    concepts::returns_reference<T> && requires(T &c) { c.pop_back(); } &&
+    !std::is_trivially_copyable_v<typename T::value_type> &&
+    (sizeof(typename T::value_type) > 32 || concepts::constructible_from_string_view<typename T::value_type>);
+
+// Removes the last element of the container on destruction while armed.
+template <typename T>
+struct pop_back_guard {
+  T &container;
+  bool armed{true};
+  ~pop_back_guard() {
+    if (armed) { container.pop_back(); }
+  }
+};
+} // namespace details
+
 /**
  * STL containers have several constructors including one that takes a single
  * size argument. Thus, some compilers (Visual Studio) will not be able to
@@ -81474,22 +102233,65 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
     SIMDJSON_TRY(val.get_array().get(arr));
   }

-  for (auto v : arr) {
-    if constexpr (concepts::returns_reference<T>) {
-      if (auto const err = v.get<value_type>().get(concepts::emplace_one(out));
-          err) {
-        // If an error occurs, the empty element that we just inserted gets
-        // removed. We're not using a temp variable because if T is a heavy
-        // type, we want the valid path to be the fast path and the slow path be
-        // the path that has errors in it.
-        if constexpr (requires { out.pop_back(); }) {
-          static_cast<void>(out.pop_back());
+  if constexpr (std::is_same_v<T, std::vector<value_type>> && !std::is_same_v<value_type, bool>) {
+    // Collect the elements in a per-thread scratch vector that keeps its
+    // capacity from call to call, then move them into out after reserving the
+    // exact size: out is allocated once instead of being regrown. A nested
+    // array of the same type finds the scratch busy and takes the paths below.
+    // Prior related work: jsonifier keeps a thread-local vector and sizes the
+    // caller's vector from that element count (parse_impl.hpp,
+    // https://github.com/nihilai-collective/Jsonifier).
+    struct scratch_space {
+      std::vector<value_type> elements{};
+      bool busy{false};
+    };
+    static thread_local scratch_space scratch;
+    if (!scratch.busy && out.empty()) {
+      struct release_scratch {
+        scratch_space &s;
+        T &out;
+        size_t parsed{0};
+        bool complete{false};
+        // On an error or an exception, out gets the elements parsed so far (as
+        // with the loops below), without allocating. Kept out of the hot path.
+        simdjson_never_inline void keep_parsed() noexcept {
+          s.elements.resize(parsed);
+          out.swap(s.elements);
         }
-        return err;
-      }
-    } else {
+        ~release_scratch() {
+          if (simdjson_unlikely(!complete)) { keep_parsed(); }
+          s.elements.clear();
+          // Do not hold on to the memory of a very large array.
+          if (s.elements.capacity() * sizeof(value_type) > (1 << 20)) { std::vector<value_type>().swap(s.elements); }
+          s.busy = false;
+        }
+      } release{scratch, out};
+      scratch.busy = true;
+      for (auto v : arr) {
+        SIMDJSON_TRY(v.get<value_type>(scratch.elements.emplace_back()));
+        release.parsed++;
+      }
+      out.reserve(release.parsed);
+      release.complete = true;
+      for (auto &e : scratch.elements) { out.emplace_back(std::move(e)); }
+      return SUCCESS;
+    }
+  }
+  if constexpr (details::deserialize_in_place<T>) {
+    for (auto v : arr) {
+      auto &slot = concepts::emplace_one(out);
+      // An error or an exception (a user tag_invoke may throw) must not leave
+      // a partially deserialized element behind.
+      details::pop_back_guard<T> guard{out};
+      SIMDJSON_TRY(v.get<value_type>(slot));
+      guard.armed = false;
+    }
+  } else {
+    for (auto v : arr) {
+      // Deserialize into a temporary first: an error or an exception (a user
+      // tag_invoke may throw) must not leave a default-constructed element behind.
       value_type temp;
-      if (auto const err = v.get<value_type>().get(temp); err) {
+      if (auto const err = v.get<value_type>(temp); err) {
         return err;
       }
       concepts::emplace_one(out, std::move(temp));
@@ -81530,7 +102332,7 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
 }

 template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, fallback::ondemand::object &obj, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, fallback::ondemand::object &obj, T &out) noexcept(false) {
   using value_type = typename std::remove_cvref_t<T>::mapped_type;

   out.clear();
@@ -81549,21 +102351,21 @@ error_code tag_invoke(deserialize_tag, fallback::ondemand::object &obj, T &out)
 }

 template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, fallback::ondemand::value &val, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, fallback::ondemand::value &val, T &out) noexcept(false) {
   fallback::ondemand::object obj;
   SIMDJSON_TRY(val.get_object().get(obj));
   return simdjson::deserialize(obj, out);
 }

 template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, fallback::ondemand::document &doc, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, fallback::ondemand::document &doc, T &out) noexcept(false) {
   fallback::ondemand::object obj;
   SIMDJSON_TRY(doc.get_object().get(obj));
   return simdjson::deserialize(obj, out);
 }

 template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, fallback::ondemand::document_reference &doc, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, fallback::ondemand::document_reference &doc, T &out) noexcept(false) {
   fallback::ondemand::object obj;
   SIMDJSON_TRY(doc.get_object().get(obj));
   return simdjson::deserialize(obj, out);
@@ -81574,10 +102376,6 @@ error_code tag_invoke(deserialize_tag, fallback::ondemand::document_reference &d
  * This CPO (Customization Point Object) will help deserialize into
  * smart pointers.
  *
- * If constructing T is nothrow, this conversion should be nothrow as well since
- * we return MEMALLOC if we're not able to allocate memory instead of throwing
- * the error message.
- *
  * @tparam T The type inside the smart pointer
  * @tparam ValT document/value type
  * @param val document/value
@@ -81585,7 +102383,7 @@ error_code tag_invoke(deserialize_tag, fallback::ondemand::document_reference &d
  * @return status of the conversion
  */
 template <concepts::smart_pointer T, typename ValT>
-error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(nothrow_deserializable<typename std::remove_cvref_t<T>::element_type, ValT>) {
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
   using element_type = typename std::remove_cvref_t<T>::element_type;

   // For better error messages, don't use these as constraints on
@@ -81597,12 +102395,13 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(nothrow_deser
       std::is_default_constructible_v<element_type>,
       "The specified type inside the unique_ptr must default constructible.");

-  auto ptr = new (std::nothrow) element_type();
-  if (ptr == nullptr) {
+  // Own the allocation before get(): a user tag_invoke may throw.
+  std::unique_ptr<element_type> ptr(new (std::nothrow) element_type());
+  if (!ptr) {
     return MEMALLOC;
   }
   SIMDJSON_TRY(val.template get<element_type>(*ptr));
-  out.reset(ptr);
+  out = std::move(ptr);
   return SUCCESS;
 }

@@ -81634,53 +102433,595 @@ error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept(nothrow_deser

 template <typename T>
 constexpr bool user_defined_type = (std::is_class_v<T>
-&& !std::is_same_v<T, std::string> && !std::is_same_v<T, std::string_view> && !concepts::optional_type<T> &&
-!concepts::appendable_containers<T>);
+&& !std::is_same_v<T, std::string> && !std::is_same_v<T, std::string_view>
+#if SIMDJSON_SUPPORTS_CHAR8_T
+// The char8_t string types are class types with no reflectable members, so
+// without this they would be taken for user structs and the reflection
+// overload below would compete with the u8 string overload in
+// std_deserialize.h, making every tag_invoke call on them ambiguous.
+&& !std::is_same_v<T, std::u8string> && !std::is_same_v<T, std::u8string_view>
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+&& !concepts::optional_type<T> &&
+!concepts::appendable_containers<T>
+// simdjson's own types (array, object, value, raw_json_string, number,
+// document, document_reference) have dedicated get<T>() specializations and
+// must never go through reflection.
+&& !is_builtin_deserializable_v<T>
+&& !std::is_same_v<T, fallback::ondemand::number>
+&& !std::is_same_v<T, fallback::ondemand::document>
+&& !std::is_same_v<T, fallback::ondemand::document_reference>);
+
+
+// key_selector_reflection_detail is defined unconditionally (it only requires
+// static reflection). It provides the compile-time machinery for building a
+// key_selector from a struct's members, the per-member helpers shared by every
+// deserialization path (they implement the annotations of annotations.h), the
+// ordered per-member path used by the opt-out build (see
+// deserialize_struct_ordered below), and the scan used for structs annotated
+// with deny_unknown_fields and as an automatic fallback (see
+// deserialize_struct_scan and keys_fit_selector below).
+namespace key_selector_reflection_detail {
+
+// A member participates if it is public, non-const, and not annotated with skip
+// or skip_deserializing.
+consteval bool is_eligible_member(std::meta::info mem) {
+  return !std::meta::is_const(mem) && std::meta::is_public(mem)
+      && !simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_tag)
+      && !simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_deserializing_tag);
+}
+
+// A field is a data member reached from T through a chain of members: usually a
+// direct member of T (a path of length one), or a member of a member annotated
+// with flatten (the members of a flattened structure are fields of the enclosing
+// one). A field is identified by the type member_path<m1, m2, ..., leaf>.
+template <std::meta::info First, std::meta::info... Rest>
+struct member_path {
+  // The data member holding the value; its annotations drive (de)serialization.
+  static constexpr std::meta::info leaf = [] {
+    std::meta::info members[] = {First, Rest...};
+    return members[sizeof...(Rest)];
+  }();
+  template <typename T>
+  static simdjson_inline constexpr auto &get(T &obj) noexcept {
+    if constexpr (sizeof...(Rest) == 0) {
+      return obj.[:First:];
+    } else {
+      return member_path<Rest...>::get(obj.[:First:]);
+    }
+  }
+};
+
+// True when T can be flattened: a structure deserialized member by member (not
+// a string, a container, an optional, a smart pointer, ...).
+template <typename T>
+constexpr bool flattenable_type = user_defined_type<T> && !concepts::string_view_keyed_map<T>
+    && !concepts::container_but_not_string<T> && !concepts::smart_pointer<T>;
+
+consteval void append_eligible_fields(std::meta::info type, std::vector<std::meta::info> &prefix,
+                                      std::vector<std::meta::info> &fields) {
+  for (std::meta::info mem : std::meta::nonstatic_data_members_of(type, std::meta::access_context::unchecked())) {
+    if (!is_eligible_member(mem)) { continue; }
+    prefix.push_back(std::meta::reflect_constant(mem));
+    if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::flatten_tag)) {
+      std::meta::info flattened = simdjson::detail::flattened_type(mem);
+      if (!std::meta::extract<bool>(std::meta::substitute(^^flattenable_type, {flattened}))) {
+        throw std::meta::exception(u8"simdjson::flatten requires a member whose type is a structure deserialized member by member", mem);
+      }
+      append_eligible_fields(flattened, prefix, fields);
+    } else {
+      fields.push_back(std::meta::substitute(^^member_path, prefix));
+    }
+    prefix.pop_back();
+  }
+}
+
+// The fields of `type` that participate in deserialization, in declaration order
+// (the fields of a flattened member take its place). A field's position in this
+// list is its "field index" below.
+consteval std::vector<std::meta::info> eligible_fields(std::meta::info type) {
+  std::vector<std::meta::info> prefix;
+  std::vector<std::meta::info> fields;
+  append_eligible_fields(type, prefix, fields);
+  return fields;
+}
+
+// The data member at the end of a field path.
+consteval std::meta::info field_leaf(std::meta::info path) {
+  return std::meta::extract<std::meta::info>(std::meta::template_arguments_of(path).back());
+}
+
+// Number of fields that participate in deserialization. A class can have zero
+// eligible fields (e.g. std::chrono::time_point, whose only data member is
+// private): an empty key_selector cannot be built, so the tag_invoke below
+// special-cases this count.
+template <typename T>
+consteval std::size_t eligible_field_count() {
+  return eligible_fields(^^T).size();
+}
+
+// The keys accepted for a member (or an enumerator): its JSON key followed by its
+// aliases. They are static strings, so they can be used as template arguments.
+consteval std::vector<const char *> accepted_keys_of(std::meta::info entity) {
+  std::vector<const char *> keys;
+  for (std::string_view key : simdjson::detail::json_key_names(entity)) {
+    bool repeated = false;
+    for (const char *previous : keys) { repeated = repeated || std::string_view(previous) == key; }
+    if (!repeated) { keys.push_back(std::define_static_string(key)); }
+  }
+  return keys;
+}
+
+// Every key accepted for T: the eligible fields in order, each followed by its
+// aliases. This is the key list of T's key_selector.
+consteval std::vector<const char *> accepted_keys(std::meta::info type) {
+  std::vector<const char *> keys;
+  for (std::meta::info path : eligible_fields(type)) {
+    for (const char *key : accepted_keys_of(field_leaf(path))) { keys.push_back(key); }
+  }
+  return keys;
+}
+
+// The field index of each entry of accepted_keys(type).
+consteval std::vector<std::size_t> accepted_key_fields(std::meta::info type) {
+  std::vector<std::size_t> key_fields;
+  std::vector<std::meta::info> fields = eligible_fields(type);
+  for (std::size_t i = 0; i < fields.size(); ++i) {
+    for (std::size_t j = 0; j < accepted_keys_of(field_leaf(fields[i])).size(); ++j) { key_fields.push_back(i); }
+  }
+  return key_fields;
+}
+
+// True when no two eligible fields of T accept the same key. Otherwise one JSON
+// key would have to fill several members: this is reported at compile time.
+template <typename T>
+consteval bool accepted_keys_are_distinct() {
+  std::vector<const char *> keys = accepted_keys(^^T);
+  for (std::size_t i = 0; i < keys.size(); ++i) {
+    for (std::size_t j = i + 1; j < keys.size(); ++j) {
+      if (std::string_view(keys[i]) == std::string_view(keys[j])) { return false; }
+    }
+  }
+  return true;
+}
+
+// True when some accepted key of T is written with escape sequences in JSON (a
+// double quote, a backslash or a control character). Such a key can only be
+// matched by comparing unescaped keys (deserialize_struct_scan): obj[key] and
+// the key_selector compare the raw bytes.
+template <typename T>
+consteval bool keys_need_unescaping() {
+  for (std::string_view key : accepted_keys(^^T)) {
+    for (char c : key) {
+      if (c == '\\' || c == '"' || static_cast<unsigned char>(c) < 0x20) { return true; }
+    }
+  }
+  return false;
+}

+// True when some eligible field of T has aliases: several selector keys may then
+// map to the same field.
+template <typename T>
+consteval bool has_aliases() {
+  return accepted_keys(^^T).size() != eligible_fields(^^T).size();
+}
+
+// True when `mem` is annotated with default_value or default_from (directly, or
+// through a default_value annotation on the enclosing structure).
+template <auto mem>
+consteval bool has_default() {
+  return simdjson::detail::has_annotation(mem, ^^simdjson::detail::default_value_tag)
+      || simdjson::detail::has_annotation(std::meta::parent_of(mem), ^^simdjson::detail::default_value_tag)
+      || simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::default_from_t) != std::meta::info{};
+}
+
+// True when a missing key for `mem` is not an error: optional members and
+// members with a default.
+template <auto mem>
+consteval bool may_be_absent() {
+  return concepts::optional_type<typename [: std::meta::type_of(mem) :]> || has_default<mem>();
+}
+
+// True when every eligible field of T is required (none may be absent). In that
+// case presence can be checked with a single match count instead of a per-field
+// "seen" array.
+template <typename T>
+consteval bool all_eligible_fields_required() {
+  bool all_required = true;
+  template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+    if constexpr (may_be_absent<[: path :]::leaf>()) {
+      all_required = false;
+    }
+  }
+  return all_required;
+}
+
+// `key` as a constevalutil::fixed_string usable as an NTTP.
+template <const char *key>
+consteval auto key_fixed_string() {
+  constexpr std::string_view key_view{ key };
+  char buffer[key_view.size() + 1] = {};
+  for (std::size_t i = 0; i < key_view.size(); ++i) { buffer[i] = key_view[i]; }
+  return constevalutil::fixed_string<key_view.size() + 1>(buffer);
+}
+
+// key_selector template arguments (one fixed_string per accepted key), in the
+// order of accepted_keys.
+template <typename T>
+consteval std::vector<std::meta::info> selector_key_args() {
+  std::vector<std::meta::info> args;
+  template for (constexpr const char *key : std::define_static_array(accepted_keys(^^T))) {
+    args.push_back(std::meta::reflect_constant(key_fixed_string<key>()));
+  }
+  return args;
+}
+
+// key_selector whose keys are exactly T's accepted keys (index i <-> i-th key).
+template <typename T>
+using selector_for = typename [: std::meta::substitute(
+    ^^fallback::ondemand::key_selector, selector_key_args<T>()) :];
+
+// True when T's accepted keys satisfy every key_selector requirement, so a
+// key_selector can be built for T without a compile-time error. This mirrors the
+// key_selector limits (see key_selector.h): at most 255 keys, each key non-empty
+// and at most 63 characters, no backslash / double-quote / null byte, and all
+// keys distinct. Keys with any other control character are excluded too: they
+// are escaped in JSON, so their raw bytes never match. When this returns false
+// the deserializer falls back to deserialize_struct_scan instead of failing to
+// compile.
+template <typename T>
+consteval bool keys_fit_selector() {
+  std::vector<std::string_view> keys;
+  for (const char *key : accepted_keys(^^T)) { keys.push_back(key); }
+  if (keys.size() > 255) { return false; }
+  for (std::size_t i = 0; i < keys.size(); ++i) {
+    if (keys[i].empty() || keys[i].size() > 63) { return false; }
+    for (char c : keys[i]) {
+      if (c == '\\' || c == '"' || static_cast<unsigned char>(c) < 0x20) { return false; }
+    }
+    for (std::size_t j = i + 1; j < keys.size(); ++j) {
+      if (keys[i] == keys[j]) { return false; }
+    }
+  }
+  return true;
+}
+
+// True when the class `adapter` declares a member named deserialize.
+consteval bool declares_deserialize(std::meta::info adapter) {
+  for (std::meta::info m : std::meta::members_of(adapter, std::meta::access_context::unchecked())) {
+    if (std::meta::has_identifier(m) && std::meta::identifier_of(m) == "deserialize") { return true; }
+  }
+  return false;
+}
+
+// Deserialize a JSON value into `target`, the storage of member `mem`, through
+// the member's with<Adapter> annotation when the adapter provides a deserialize
+// function.
+template <auto mem, typename ValueT, typename M>
+simdjson_warn_unused simdjson_inline error_code deserialize_member_value(ValueT &field_value, M &target) noexcept(false) {
+  constexpr std::meta::info with_type = simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::with_t);
+  if constexpr (with_type != std::meta::info{}) {
+    using adapter = typename [: with_type :]::adapter;
+    using ondemand_value = fallback::ondemand::value;
+    if constexpr (requires { { adapter::deserialize(field_value, target) } -> std::convertible_to<error_code>; }) {
+      return adapter::deserialize(field_value, target);
+    } else if constexpr (requires(ondemand_value &v) { { adapter::deserialize(v, target) } -> std::convertible_to<error_code>; }
+                         && requires { field_value.get_value(); }) {
+      // A transparent structure read from a document: the adapter takes an
+      // ondemand::value. A scalar document cannot be viewed as a value, so it
+      // reports SCALAR_DOCUMENT_AS_VALUE (an adapter taking auto& receives the
+      // document itself and has no such limitation).
+      ondemand_value v;
+      SIMDJSON_TRY(field_value.get_value().get(v));
+      return adapter::deserialize(v, target);
+    } else {
+      static_assert(!declares_deserialize(^^adapter),
+                    "the deserialize function of a simdjson::with adapter must be callable as "
+                    "Adapter::deserialize(simdjson::ondemand::value &, T &) and return an error_code");
+      return field_value.get(target);
+    }
+  } else {
+    return field_value.get(target);
+  }
+}
+
+// Deserialize the storage `target` of member `mem` from a JSON value.
+template <auto mem, typename ValueT, typename M>
+simdjson_warn_unused simdjson_inline error_code deserialize_member(ValueT &field_value, M &target) noexcept(false) {
+  if constexpr (has_default<mem>() && std::is_default_constructible_v<M> && std::is_move_assignable_v<M>) {
+    // A present key replaces the default value: deserialize into a fresh
+    // temporary so that, e.g., a container does not append to its default
+    // content, and a failure leaves the default untouched.
+    M value{};
+    SIMDJSON_TRY(deserialize_member_value<mem>(field_value, value));
+    target = std::move(value);
+    return SUCCESS;
+  } else {
+    return deserialize_member_value<mem>(field_value, target);
+  }
+}

+// Deserialize the field with the given field index.
+template <typename T>
+simdjson_warn_unused simdjson_inline error_code deserialize_field_at(
+    std::size_t field_index, fallback::ondemand::value field_value, T &out) noexcept(false) {
+  std::size_t counter = 0;
+  template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+    using field = [: path :];
+    if (field_index == counter) { return deserialize_member<field::leaf>(field_value, field::get(out)); }
+    ++counter;
+  }
+  return SUCCESS;
+}
+
+// Called when the key(s) of member `mem`, stored in `target`, are missing from
+// the JSON object: an error for a required member, a call to the default_from
+// factory, or nothing at all.
+template <auto mem, typename M>
+simdjson_warn_unused simdjson_inline error_code on_missing_member(M &target) noexcept(false) {
+  constexpr std::meta::info default_from_type = simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::default_from_t);
+  if constexpr (default_from_type != std::meta::info{}) {
+    target = [: default_from_type :]::factory();
+    return SUCCESS;
+  } else if constexpr (may_be_absent<mem>()) {
+    // For optional and default_value members, a missing key is not an error:
+    // leave the member at its current (default) value.
+    (void)target;
+    return SUCCESS;
+  } else {
+    (void)target;
+    return NO_SUCH_FIELD;
+  }
+}
+
+// Report or handle every field that was not seen in the JSON object.
+template <typename T, std::size_t N>
+simdjson_warn_unused simdjson_inline error_code handle_missing_fields(
+    const std::array<bool, N> &seen_field, T &out) noexcept(false) {
+  std::size_t counter = 0;
+  template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+    using field = [: path :];
+    if (!seen_field[counter]) { SIMDJSON_TRY(on_missing_member<field::leaf>(field::get(out))); }
+    ++counter;
+  }
+  return SUCCESS;
+}
+
+// Ordered, per-field deserialization: one obj[key] lookup per eligible field
+// (and per alias, until one is found). This is the opt-out path
+// (-DSIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION=1), except for structs with keys
+// that need unescaping (see keys_need_unescaping).
+template <typename T>
+simdjson_warn_unused error_code deserialize_struct_ordered(
+    fallback::ondemand::object &obj, T &out) noexcept(false) {
+  template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+    using field = [: path :];
+    fallback::ondemand::value field_value;
+    error_code error = NO_SUCH_FIELD;
+    template for (constexpr const char *key : std::define_static_array(accepted_keys_of(field::leaf))) {
+      if (error == NO_SUCH_FIELD) { error = obj[std::string_view(key)].get(field_value); }
+    }
+    if (error == NO_SUCH_FIELD) {
+      SIMDJSON_TRY(on_missing_member<field::leaf>(field::get(out)));
+    } else if (error) {
+      return error;
+    } else {
+      SIMDJSON_TRY(deserialize_member<field::leaf>(field_value, field::get(out)));
+    }
+  }
+  return SUCCESS;
+}
+
+// Appends the JSON keys that serialization writes for members of `type` that
+// deserialization cannot assign (const or non-public members, directly or
+// through flatten). `all` is true inside a flattened member that is itself
+// unassignable. Keys of skip_deserializing members are not included: they are
+// unknown keys, as documented.
+consteval void append_unassignable_keys(std::meta::info type, bool all, std::vector<const char *> &keys) {
+  for (std::meta::info mem : std::meta::nonstatic_data_members_of(type, std::meta::access_context::unchecked())) {
+    if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_tag)
+        || simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_serializing_tag)
+        || simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_deserializing_tag)) {
+      continue;
+    }
+    bool unassignable = all || !is_eligible_member(mem);
+    if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::flatten_tag)) {
+      append_unassignable_keys(simdjson::detail::flattened_type(mem), unassignable, keys);
+    } else if (unassignable) {
+      keys.push_back(std::define_static_string(simdjson::detail::json_key_name(mem)));
+    }
+  }
+}
+
+consteval std::vector<const char *> unassignable_keys(std::meta::info type) {
+  std::vector<const char *> keys;
+  append_unassignable_keys(type, false, keys);
+  return keys;
+}
+
+// Deserialization by a single pass over every field of the object, comparing
+// unescaped keys. It is used for structs annotated with deny_unknown_fields
+// (DenyUnknown = true), where a key that does not map to an eligible field is
+// reported as UNKNOWN_FIELD, and as the fallback for structs whose keys the
+// key_selector or obj[key] cannot match (see keys_fit_selector and
+// keys_need_unescaping). As with object::for_each, the first occurrence of a
+// field wins (later duplicates, or aliases of a field already seen, are ignored).
+template <bool DenyUnknown, typename T>
+simdjson_warn_unused error_code deserialize_struct_scan(
+    fallback::ondemand::object &obj, T &out) noexcept(false) {
+  static constexpr auto keys = std::define_static_array(accepted_keys(^^T));
+  static constexpr auto key_fields = std::define_static_array(accepted_key_fields(^^T));
+  std::array<bool, eligible_field_count<T>()> seen_field{};
+  for (auto field_result : obj) {
+    fallback::ondemand::field json_field;
+    SIMDJSON_TRY(std::move(field_result).get(json_field));
+    std::string_view key;
+    SIMDJSON_TRY(json_field.unescaped_key().get(key));
+    std::size_t key_index = keys.size();
+    for (std::size_t i = 0; i < keys.size(); ++i) {
+      if (key == std::string_view(keys[i])) { key_index = i; break; }
+    }
+    if (key_index == keys.size()) {
+      if constexpr (DenyUnknown) {
+        // A key that T itself serializes (e.g. of a const member) is not
+        // unknown: a serialized value must parse back.
+        static constexpr auto ignored_keys = std::define_static_array(unassignable_keys(^^T));
+        bool ignored = false;
+        for (const char *ignored_key : ignored_keys) {
+          if (key == std::string_view(ignored_key)) { ignored = true; break; }
+        }
+        if (!ignored) { return UNKNOWN_FIELD; }
+      }
+      continue;
+    }
+    const std::size_t field_index = key_fields[key_index];
+    if (seen_field[field_index]) { continue; }
+    seen_field[field_index] = true;
+    SIMDJSON_TRY(deserialize_field_at(field_index, json_field.value(), out));
+  }
+  return handle_missing_fields(seen_field, out);
+}
+
+template <typename T>
+consteval bool is_transparent() {
+  return simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag);
+}
+
+} // namespace key_selector_reflection_detail
+
+// Deserialize a reflected struct. By default this builds a compile-time
+// key_selector from the struct's members and walks each object once with
+// object::for_each (perfect-hash key matching), instead of one obj[key] lookup
+// per member. Other paths are used instead:
+//   - globally, the ordered per-member path when defining
+//     -DSIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION=1;
+//   - automatically and per-type, a scan of the object comparing unescaped keys
+//     when the struct's keys do not fit the key_selector limits (see
+//     keys_fit_selector) or cannot be compared raw (see keys_need_unescaping),
+//     so that long member names and the like keep compiling rather than
+//     tripping a static_assert.
+// Structs annotated with deny_unknown_fields always use the strict scan, and
+// structs annotated with transparent are deserialized as their single member.
+//
+// noexcept(false): a member's tag_invoke may throw and the exception must reach
+// the caller of get<T>().
 template <typename T, typename ValT>
   requires(user_defined_type<T> && std::is_class_v<T>)
-error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
+  if constexpr (key_selector_reflection_detail::is_transparent<T>()) {
+    constexpr auto mem = simdjson::detail::transparent_member(^^T);
+    if constexpr (std::is_same_v<std::remove_cvref_t<ValT>, fallback::ondemand::object>) {
+      // We were handed an object: only a structure can be deserialized from it.
+      if constexpr (user_defined_type<typename [: std::meta::type_of(mem) :]>) {
+        return tag_invoke(deserialize_tag{}, val, out.[:mem:]);
+      } else {
+        return INCORRECT_TYPE;
+      }
+    } else {
+      return key_selector_reflection_detail::deserialize_member<mem>(val, out.[:mem:]);
+    }
+  } else {
+  static_assert(key_selector_reflection_detail::accepted_keys_are_distinct<T>(),
+                "two members of this structure accept the same JSON key (check rename, alias, "
+                "rename_all and flatten)");
   fallback::ondemand::object obj;
   if constexpr (std::is_same_v<std::remove_cvref_t<ValT>, fallback::ondemand::object>) {
     obj = val;
   } else {
     SIMDJSON_TRY(val.get_object().get(obj));
   }
-  template for (constexpr auto mem : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
-    if constexpr (!std::meta::is_const(mem) && std::meta::is_public(mem)) {
-      constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
-      if constexpr (concepts::optional_type<decltype(out.[:mem:])>) {
-        // for optional members, it's ok if the key is missing
-        auto error = obj[key].get(out.[:mem:]);
-        if (error && error != NO_SUCH_FIELD) {
-          if(error == NO_SUCH_FIELD) {
-            out.[:mem:].reset();
-            continue;
-          }
-          return error;
-        }
-      } else {
-        // for non-optional members, the key must be present
-        SIMDJSON_TRY(obj[key].get(out.[:mem:]));
+  if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::deny_unknown_fields_tag)) {
+    return key_selector_reflection_detail::deserialize_struct_scan<true>(obj, out);
+  } else {
+#if defined(SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION) && SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION
+  // Opt-out build: use the ordered per-member path, unless obj[key] cannot
+  // match T's keys.
+  if constexpr (key_selector_reflection_detail::keys_need_unescaping<T>()) {
+    return key_selector_reflection_detail::deserialize_struct_scan<false>(obj, out);
+  } else {
+    return key_selector_reflection_detail::deserialize_struct_ordered(obj, out);
+  }
+#else
+  if constexpr (key_selector_reflection_detail::eligible_field_count<T>() == 0) {
+    // No fields to deserialize: an empty key_selector cannot be built, so just
+    // validate that the input is an object (done above) and succeed. Mirrors the
+    // ordered per-member path, which iterates over zero members.
+    (void)out;
+    (void)obj;
+    return SUCCESS;
+  } else if constexpr (!key_selector_reflection_detail::keys_fit_selector<T>()) {
+    // Automatic fallback: T's accepted keys do not fit the key_selector limits
+    // (e.g. a member name longer than 63 characters, or a key with a double
+    // quote), so building a selector would be a compile error. Scan the object
+    // instead, so the default never breaks a struct that the opt-out path would
+    // accept.
+    return key_selector_reflection_detail::deserialize_struct_scan<false>(obj, out);
+  } else {
+  using selector = key_selector_reflection_detail::selector_for<T>;
+  if constexpr (key_selector_reflection_detail::all_eligible_fields_required<T>()
+                && !key_selector_reflection_detail::has_aliases<T>()) {
+    // Fast path: every member is required and has a single key. A single
+    // for_each pass parses each matched field; the returned match count then
+    // tells us whether every member was present (matched_count ==
+    // selector::size()) without a per-member "seen" array. A value-parse error
+    // (e.g. a type mismatch) is propagated by for_each.
+    auto walk = obj.template for_each<selector>(
+        [&](std::size_t matched_index, fallback::ondemand::value field_value) -> error_code {
+      std::size_t counter = 0;
+      template for (constexpr auto path : std::define_static_array(key_selector_reflection_detail::eligible_fields(^^T))) {
+        using field = [: path :];
+        if (matched_index == counter) { return key_selector_reflection_detail::deserialize_member<field::leaf>(field_value, field::get(out)); }
+        ++counter;
       }
-    }
-  };
-  return simdjson::SUCCESS;
-}
-
-// Support for enum deserialization - deserialize from string representation using expand approach from P2996R12
+      return SUCCESS;
+    });
+    if (walk.error) { return walk.error; }
+    // A missing required member shows up as a short match count and is reported as
+    // NO_SUCH_FIELD, mirroring the ordered obj[key] path.
+    if (walk.matched_count != selector::size()) { return NO_SUCH_FIELD; }
+    return SUCCESS;
+  } else {
+    static constexpr auto key_fields = std::define_static_array(key_selector_reflection_detail::accepted_key_fields(^^T));
+    std::array<bool, key_selector_reflection_detail::eligible_field_count<T>()> seen_field{};
+    // Single pass over the object: each field whose key matches a member (or one
+    // of its aliases) yields its selector index, which we map back to the
+    // corresponding member. The first key seen for a member wins. The callback
+    // returns an error_code so that a value-parse error (e.g. a type mismatch on
+    // a matched field) is propagated by for_each instead of being silently dropped.
+    error_code walk_error = obj.template for_each<selector>(
+        [&](std::size_t matched_index, fallback::ondemand::value field_value) -> error_code {
+      const std::size_t field_index = key_fields[matched_index];
+      if (seen_field[field_index]) { return SUCCESS; }
+      seen_field[field_index] = true;
+      return key_selector_reflection_detail::deserialize_field_at(field_index, field_value, out);
+    });
+    if (walk_error) { return walk_error; }
+    // Required members must be present: a missing one is reported as
+    // NO_SUCH_FIELD, mirroring the ordered obj[key] path. Optional and defaulted
+    // members may be absent.
+    return key_selector_reflection_detail::handle_missing_fields(seen_field, out);
+  }
+  }
+#endif // SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION
+  }
+  }
+}
+
+// Support for enum deserialization - deserialize from string representation using
+// expand approach from P2996R12. The accepted strings are the enumerator's JSON key
+// (see rename and rename_all) and its aliases.
 template <typename T, typename ValT>
   requires(std::is_enum_v<T>)
 error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
 #if SIMDJSON_STATIC_REFLECTION
   std::string_view str;
   SIMDJSON_TRY(val.get_string().get(str));
-  constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
+  static constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
   template for (constexpr auto enum_val : enumerators) {
-    if (str == std::meta::identifier_of(enum_val)) {
-      out = [:enum_val:];
-      return SUCCESS;
+    template for (constexpr const char *key : std::define_static_array(key_selector_reflection_detail::accepted_keys_of(enum_val))) {
+      if (str == std::string_view(key)) {
+        out = [:enum_val:];
+        return SUCCESS;
+      }
     }
   };

@@ -81696,33 +103037,25 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {

 template <typename simdjson_value, typename T>
   requires(user_defined_type<std::remove_cvref_t<T>>)
-error_code tag_invoke(deserialize_tag, simdjson_value &val, std::unique_ptr<T> &out) noexcept {
-  if (!out) {
-    out = std::make_unique<T>();
-    if (!out) {
-      return MEMALLOC;
-    }
-  }
-  if (auto err = val.get(*out)) {
-    out.reset();
-    return err;
+error_code tag_invoke(deserialize_tag, simdjson_value &val, std::unique_ptr<T> &out) noexcept(false) {
+  std::unique_ptr<T> ptr(new (std::nothrow) T());
+  if (!ptr) {
+    return MEMALLOC;
   }
+  SIMDJSON_TRY(val.get(*ptr));
+  out = std::move(ptr);
   return SUCCESS;
 }

 template <typename simdjson_value, typename T>
   requires(user_defined_type<std::remove_cvref_t<T>>)
-error_code tag_invoke(deserialize_tag, simdjson_value &val, std::shared_ptr<T> &out) noexcept {
-  if (!out) {
-    out = std::make_shared<T>();
-    if (!out) {
-      return MEMALLOC;
-    }
-  }
-  if (auto err = val.get(*out)) {
-    out.reset();
-    return err;
+error_code tag_invoke(deserialize_tag, simdjson_value &val, std::shared_ptr<T> &out) noexcept(false) {
+  std::shared_ptr<T> ptr(new (std::nothrow) T());
+  if (!ptr) {
+    return MEMALLOC;
   }
+  SIMDJSON_TRY(val.get(*ptr));
+  out = std::move(ptr);
   return SUCCESS;
 }

@@ -82034,9 +103367,17 @@ simdjson_inline simdjson_result<array> array::started(value_iterator &iter) noex
   return array(iter);
 }

-simdjson_inline simdjson_result<array_iterator> array::begin() noexcept {
+simdjson_inline simdjson_result<array_iterator> array::begin() & noexcept {
 #if SIMDJSON_DEVELOPMENT_CHECKS
-  if (!iter.is_at_iterator_start()) { return OUT_OF_ORDER_ITERATION; }
+  if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
+  return array_iterator(iter, this);
+#endif
+  return array_iterator(iter);
+}
+simdjson_inline simdjson_result<array_iterator> array::begin() && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  // The array is a temporary that the iterator may outlive: do not lock it.
+  if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
 #endif
   return array_iterator(iter);
 }
@@ -82063,6 +103404,9 @@ simdjson_inline simdjson_result<std::string_view> array::raw_json() noexcept {
 SIMDJSON_PUSH_DISABLE_WARNINGS
 SIMDJSON_DISABLE_STRICT_OVERFLOW_WARNING
 simdjson_inline simdjson_result<size_t> array::count_elements() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   size_t count{0};
   // Important: we do not consume any of the values.
   for(simdjson_unused auto v : *this) { count++; }
@@ -82076,6 +103420,9 @@ simdjson_inline simdjson_result<size_t> array::count_elements() & noexcept {
 SIMDJSON_POP_DISABLE_WARNINGS

 simdjson_inline simdjson_result<bool> array::is_empty() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   bool is_not_empty;
   auto error = iter.reset_array().get(is_not_empty);
   if(error) { return error; }
@@ -82083,31 +103430,30 @@ simdjson_inline simdjson_result<bool> array::is_empty() & noexcept {
 }

 inline simdjson_result<bool> array::reset() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   return iter.reset_array();
 }

+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline void array::set_locked(bool _locked) noexcept {
+  locked = _locked;
+}
+#endif
+
 inline simdjson_result<value> array::at_pointer(std::string_view json_pointer) noexcept {
-  if (json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
+  // An empty pointer has no json_pointer[0]: with a default-constructed
+  // std::string_view, reading it dereferences a null pointer.
+  if (json_pointer.empty() || json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
   json_pointer = json_pointer.substr(1);
   // - means "the append position" or "the element after the end of the array"
   // We don't support this, because we're returning a real element, not a position.
   if (json_pointer == "-") { return INDEX_OUT_OF_BOUNDS; }

-  // Read the array index
   size_t array_index = 0;
   size_t i;
-  for (i = 0; i < json_pointer.length() && json_pointer[i] != '/'; i++) {
-    uint8_t digit = uint8_t(json_pointer[i] - '0');
-    // Check for non-digit in array index. If it's there, we're trying to get a field in an object
-    if (digit > 9) { return INCORRECT_TYPE; }
-    array_index = array_index*10 + digit;
-  }
-
-  // 0 followed by other digits is invalid
-  if (i > 1 && json_pointer[0] == '0') { return INVALID_JSON_POINTER; } // "JSON pointer array index has other characters after 0"
-
-  // Empty string is invalid; so is a "/" with no digits before it
-  if (i == 0) { return INVALID_JSON_POINTER; } // "Empty string in JSON pointer array index"
+  SIMDJSON_TRY(internal::parse_json_pointer_array_index(json_pointer, array_index, i));
   // Get the child
   auto child = at(array_index);
   // If there is an error, it ends here
@@ -82181,6 +103527,9 @@ inline error_code array::for_each_at_path_with_wildcard(std::string_view json_pa
 }

 simdjson_inline simdjson_result<value> array::at(size_t index) noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   size_t i = 0;
   for (auto value : *this) {
     if (i == index) { return value; }
@@ -82210,10 +103559,14 @@ simdjson_inline simdjson_result<fallback::ondemand::array>::simdjson_result(
 {
 }

-simdjson_inline simdjson_result<fallback::ondemand::array_iterator> simdjson_result<fallback::ondemand::array>::begin() noexcept {
+simdjson_inline simdjson_result<fallback::ondemand::array_iterator> simdjson_result<fallback::ondemand::array>::begin() & noexcept {
   if (error()) { return error(); }
   return first.begin();
 }
+simdjson_inline simdjson_result<fallback::ondemand::array_iterator> simdjson_result<fallback::ondemand::array>::begin() && noexcept {
+  if (error()) { return error(); }
+  return std::move(first).begin();
+}
 simdjson_inline simdjson_result<fallback::ondemand::array_iterator> simdjson_result<fallback::ondemand::array>::end() noexcept {
   if (error()) { return error(); }
   return first.end();
@@ -82276,6 +103629,59 @@ simdjson_inline array_iterator::array_iterator(const value_iterator &_iter) noex
   : iter{_iter}
 {}

+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline array_iterator::array_iterator(const value_iterator &_iter, array* _parent) noexcept
+  : parent{_parent}, iter{_iter}
+{
+  if (parent) parent->set_locked(true);
+}
+
+simdjson_inline array_iterator::~array_iterator() noexcept
+{
+  if (parent) parent->set_locked(false);
+}
+
+simdjson_inline array_iterator::array_iterator(array_iterator&& other) noexcept
+  : has_been_referenced{other.has_been_referenced},
+    parent{other.parent},
+    iter{std::move(other.iter)}
+{
+  other.parent = nullptr;
+}
+
+simdjson_inline array_iterator& array_iterator::operator=(array_iterator&& other) noexcept {
+  if (this != &other)
+  {
+    if (parent)
+      parent->set_locked(false);
+    has_been_referenced = other.has_been_referenced;
+    parent = other.parent;
+    iter = std::move(other.iter);
+
+    other.parent = nullptr;
+  }
+  return *this;
+}
+
+simdjson_inline array_iterator::array_iterator(const array_iterator& other) noexcept
+  : has_been_referenced{other.has_been_referenced},
+    parent{nullptr},
+    iter{other.iter}
+{}
+
+simdjson_inline array_iterator& array_iterator::operator=(const array_iterator& other) noexcept {
+  if (this != &other)
+  {
+    if (parent)
+      parent->set_locked(false);
+    has_been_referenced = other.has_been_referenced;
+    parent = nullptr;
+    iter = other.iter;
+  }
+  return *this;
+}
+#endif
+
 simdjson_inline simdjson_result<value> array_iterator::operator*() noexcept {
 #if SIMDJSON_DEVELOPMENT_CHECKS
    SIMDJSON_ASSUME(!has_been_referenced);
@@ -82371,6 +103777,41 @@ namespace simdjson {
 namespace fallback {
 namespace ondemand {

+/**
+ * @private Narrows the result of get_uint64() to a smaller unsigned integer type T.
+ * @returns NUMBER_OUT_OF_RANGE if the value does not fit in a T.
+ */
+template <typename T>
+simdjson_inline simdjson_result<T> narrow_integer(simdjson_result<uint64_t> wide) noexcept {
+  uint64_t result;
+  SIMDJSON_TRY(std::move(wide).get(result));
+  if (result > static_cast<uint64_t>((std::numeric_limits<T>::max)())) { return NUMBER_OUT_OF_RANGE; }
+  return static_cast<T>(result);
+}
+
+/**
+ * @private Narrows the result of get_int64() to a smaller signed integer type T.
+ * @returns NUMBER_OUT_OF_RANGE if the value does not fit in a T.
+ */
+template <typename T>
+simdjson_inline simdjson_result<T> narrow_integer(simdjson_result<int64_t> wide) noexcept {
+  int64_t result;
+  SIMDJSON_TRY(std::move(wide).get(result));
+  if (result > static_cast<int64_t>((std::numeric_limits<T>::max)()) || result < static_cast<int64_t>((std::numeric_limits<T>::min)())) { return NUMBER_OUT_OF_RANGE; }
+  return static_cast<T>(result);
+}
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+// get_float32() parses with get_float(), so float must be binary32.
+static_assert(std::numeric_limits<float>::is_iec559 && std::numeric_limits<float>::digits == 24,
+              "get_float32() requires float to be IEEE binary32");
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+// get_float64() parses with get_double(), so double must be binary64.
+static_assert(std::numeric_limits<double>::is_iec559 && std::numeric_limits<double>::digits == 53,
+              "get_float64() requires double to be IEEE binary64");
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
+
 simdjson_inline value::value(const value_iterator &_iter) noexcept
   : iter{_iter}
 {
@@ -82402,6 +103843,13 @@ simdjson_inline simdjson_result<raw_json_string> value::get_raw_json_string() no
 simdjson_inline simdjson_result<std::string_view> value::get_string(bool allow_replacement) noexcept {
   return iter.get_string(allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> value::get_u8string(bool allow_replacement) noexcept {
+  std::string_view content;
+  SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+  return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code value::get_string(string_type& receiver, bool allow_replacement) noexcept {
   return iter.get_string(receiver, allow_replacement);
@@ -82415,6 +103863,12 @@ simdjson_inline simdjson_result<double> value::get_double() noexcept {
 simdjson_inline simdjson_result<double> value::get_double_in_string() noexcept {
   return iter.get_double_in_string();
 }
+simdjson_inline simdjson_result<float> value::get_float() noexcept {
+  return iter.get_float();
+}
+simdjson_inline simdjson_result<float> value::get_float_in_string() noexcept {
+  return iter.get_float_in_string();
+}
 simdjson_inline simdjson_result<uint64_t> value::get_uint64() noexcept {
   return iter.get_uint64();
 }
@@ -82428,17 +103882,37 @@ simdjson_inline simdjson_result<int64_t> value::get_int64_in_string() noexcept {
   return iter.get_int64_in_string();
 }
 simdjson_inline simdjson_result<uint32_t> value::get_uint32() noexcept {
-  uint64_t result;
-  SIMDJSON_TRY(get_uint64().get(result));
-  if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<uint32_t>(result);
+  return narrow_integer<uint32_t>(get_uint64());
 }
 simdjson_inline simdjson_result<int32_t> value::get_int32() noexcept {
-  int64_t result;
-  SIMDJSON_TRY(get_int64().get(result));
-  if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<int32_t>(result);
+  return narrow_integer<int32_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint16_t> value::get_uint16() noexcept {
+  return narrow_integer<uint16_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int16_t> value::get_int16() noexcept {
+  return narrow_integer<int16_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint8_t> value::get_uint8() noexcept {
+  return narrow_integer<uint8_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int8_t> value::get_int8() noexcept {
+  return narrow_integer<int8_t>(get_int64());
 }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> value::get_float32() noexcept {
+  float result;
+  SIMDJSON_TRY(get_float().get(result));
+  return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> value::get_float64() noexcept {
+  double result;
+  SIMDJSON_TRY(get_double().get(result));
+  return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<bool> value::get_bool() noexcept {
   return iter.get_bool();
 }
@@ -82450,12 +103924,26 @@ template<> simdjson_inline simdjson_result<array> value::get() noexcept { return
 template<> simdjson_inline simdjson_result<object> value::get() noexcept { return get_object(); }
 template<> simdjson_inline simdjson_result<raw_json_string> value::get() noexcept { return get_raw_json_string(); }
 template<> simdjson_inline simdjson_result<std::string_view> value::get() noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> value::get() noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_inline simdjson_result<number> value::get() noexcept { return get_number(); }
 template<> simdjson_inline simdjson_result<double> value::get() noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> value::get() noexcept { return get_float(); }
 template<> simdjson_inline simdjson_result<uint64_t> value::get() noexcept { return get_uint64(); }
 template<> simdjson_inline simdjson_result<int64_t> value::get() noexcept { return get_int64(); }
 template<> simdjson_inline simdjson_result<uint32_t> value::get() noexcept { return get_uint32(); }
 template<> simdjson_inline simdjson_result<int32_t> value::get() noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> value::get() noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> value::get() noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> value::get() noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> value::get() noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> value::get() noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> value::get() noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<> simdjson_inline simdjson_result<bool> value::get() noexcept { return get_bool(); }


@@ -82463,12 +103951,26 @@ template<> simdjson_warn_unused simdjson_inline error_code value::get(array& out
 template<> simdjson_warn_unused simdjson_inline error_code value::get(object& out) noexcept { return get_object().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(raw_json_string& out) noexcept { return get_raw_json_string().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(std::string_view& out) noexcept { return get_string(false).get(out); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::u8string_view& out) noexcept { return get_u8string(false).get(out); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_warn_unused simdjson_inline error_code value::get(number& out) noexcept { return get_number().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(double& out) noexcept { return get_double().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(float& out) noexcept { return get_float().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(uint64_t& out) noexcept { return get_uint64().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(int64_t& out) noexcept { return get_int64().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(uint32_t& out) noexcept { return get_uint32().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(int32_t& out) noexcept { return get_int32().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(uint16_t& out) noexcept { return get_uint16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(int16_t& out) noexcept { return get_int16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(uint8_t& out) noexcept { return get_uint8().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(int8_t& out) noexcept { return get_int8().get(out); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::float32_t& out) noexcept { return get_float32().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::float64_t& out) noexcept { return get_float64().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<>  simdjson_warn_unused simdjson_inline error_code value::get(bool& out) noexcept { return get_bool().get(out); }

 #if SIMDJSON_EXCEPTIONS
@@ -82637,6 +104139,9 @@ inline bool is_pointer_well_formed(std::string_view json_pointer) noexcept {
 }

 simdjson_inline simdjson_result<value> value::at_pointer(std::string_view json_pointer) noexcept {
+  // The empty JSON Pointer refers to the whole value (RFC 6901), as in
+  // document::at_pointer.
+  if (json_pointer.empty()) { return value(iter); }
   json_type t;
   SIMDJSON_TRY(type().get(t));
   switch (t)
@@ -82674,6 +104179,10 @@ template <typename Func>
 template <typename Func>
 #endif
 inline error_code value::for_each_at_path_with_wildcard(std::string_view json_path, Func&& callback) noexcept {
+  // Every recursive step of for_each_at_path_with_wildcard goes through this
+  // function, and each one descends one level into the document. A path with
+  // many segments applied to a deeply nested document would otherwise recurse
+  // without bound and overflow the stack.
   if (size_t(iter.depth()) >= iter.json_iter().parser->max_depth()) { return DEPTH_ERROR; }
   json_type t;
   SIMDJSON_TRY(type().get(t));
@@ -82787,10 +104296,46 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<fallback::ondemand::val
   if (error()) { return error(); }
   return first.get_int32();
 }
+simdjson_inline simdjson_result<uint16_t> simdjson_result<fallback::ondemand::value>::get_uint16() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<fallback::ondemand::value>::get_int16() noexcept {
+  if (error()) { return error(); }
+  return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<fallback::ondemand::value>::get_uint8() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<fallback::ondemand::value>::get_int8() noexcept {
+  if (error()) { return error(); }
+  return first.get_int8();
+}
 simdjson_inline simdjson_result<double> simdjson_result<fallback::ondemand::value>::get_double() noexcept {
   if (error()) { return error(); }
   return first.get_double();
 }
+simdjson_inline simdjson_result<float> simdjson_result<fallback::ondemand::value>::get_float() noexcept {
+  if (error()) { return error(); }
+  return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<fallback::ondemand::value>::get_float_in_string() noexcept {
+  if (error()) { return error(); }
+  return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<fallback::ondemand::value>::get_float32() noexcept {
+  if (error()) { return error(); }
+  return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<fallback::ondemand::value>::get_float64() noexcept {
+  if (error()) { return error(); }
+  return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<double> simdjson_result<fallback::ondemand::value>::get_double_in_string() noexcept {
   if (error()) { return error(); }
   return first.get_double_in_string();
@@ -82799,6 +104344,12 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<fallback::onde
   if (error()) { return error(); }
   return first.get_string(allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<fallback::ondemand::value>::get_u8string(bool allow_replacement) noexcept {
+  if (error()) { return error(); }
+  return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_inline error_code simdjson_result<fallback::ondemand::value>::get_string(string_type& receiver, bool allow_replacement) noexcept {
   if (error()) { return error(); }
@@ -82827,11 +104378,23 @@ template<> simdjson_inline error_code simdjson_result<fallback::ondemand::value>
   return SUCCESS;
 }

-template<typename T> simdjson_inline simdjson_result<T> simdjson_result<fallback::ondemand::value>::get() noexcept {
+template<typename T> simdjson_inline simdjson_result<T> simdjson_result<fallback::ondemand::value>::get()
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, fallback::ondemand::value>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>();
 }
-template<typename T> simdjson_inline error_code simdjson_result<fallback::ondemand::value>::get(T &out) noexcept {
+template<typename T> simdjson_inline error_code simdjson_result<fallback::ondemand::value>::get(T &out)
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, fallback::ondemand::value>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>(out);
 }
@@ -83101,16 +104664,22 @@ simdjson_inline simdjson_result<int64_t> document::get_int64_in_string() noexcep
   return get_root_value_iterator().get_root_int64_in_string(true);
 }
 simdjson_inline simdjson_result<uint32_t> document::get_uint32() noexcept {
-  uint64_t result;
-  SIMDJSON_TRY(get_uint64().get(result));
-  if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<uint32_t>(result);
+  return narrow_integer<uint32_t>(get_uint64());
 }
 simdjson_inline simdjson_result<int32_t> document::get_int32() noexcept {
-  int64_t result;
-  SIMDJSON_TRY(get_int64().get(result));
-  if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<int32_t>(result);
+  return narrow_integer<int32_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint16_t> document::get_uint16() noexcept {
+  return narrow_integer<uint16_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int16_t> document::get_int16() noexcept {
+  return narrow_integer<int16_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint8_t> document::get_uint8() noexcept {
+  return narrow_integer<uint8_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int8_t> document::get_int8() noexcept {
+  return narrow_integer<int8_t>(get_int64());
 }
 simdjson_inline simdjson_result<double> document::get_double() noexcept {
   return get_root_value_iterator().get_root_double(true);
@@ -83118,9 +104687,36 @@ simdjson_inline simdjson_result<double> document::get_double() noexcept {
 simdjson_inline simdjson_result<double> document::get_double_in_string() noexcept {
   return get_root_value_iterator().get_root_double_in_string(true);
 }
+simdjson_inline simdjson_result<float> document::get_float() noexcept {
+  return get_root_value_iterator().get_root_float(true);
+}
+simdjson_inline simdjson_result<float> document::get_float_in_string() noexcept {
+  return get_root_value_iterator().get_root_float_in_string(true);
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> document::get_float32() noexcept {
+  float result;
+  SIMDJSON_TRY(get_float().get(result));
+  return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> document::get_float64() noexcept {
+  double result;
+  SIMDJSON_TRY(get_double().get(result));
+  return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<std::string_view> document::get_string(bool allow_replacement) noexcept {
   return get_root_value_iterator().get_root_string(true, allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> document::get_u8string(bool allow_replacement) noexcept {
+  std::string_view content;
+  SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+  return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code document::get_string(string_type& receiver, bool allow_replacement) noexcept {
   return get_root_value_iterator().get_root_string(receiver, true, allow_replacement);
@@ -83142,11 +104738,25 @@ template<> simdjson_inline simdjson_result<array> document::get() & noexcept { r
 template<> simdjson_inline simdjson_result<object> document::get() & noexcept { return get_object(); }
 template<> simdjson_inline simdjson_result<raw_json_string> document::get() & noexcept { return get_raw_json_string(); }
 template<> simdjson_inline simdjson_result<std::string_view> document::get() & noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> document::get() & noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_inline simdjson_result<double> document::get() & noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> document::get() & noexcept { return get_float(); }
 template<> simdjson_inline simdjson_result<uint64_t> document::get() & noexcept { return get_uint64(); }
 template<> simdjson_inline simdjson_result<int64_t> document::get() & noexcept { return get_int64(); }
 template<> simdjson_inline simdjson_result<uint32_t> document::get() & noexcept { return get_uint32(); }
 template<> simdjson_inline simdjson_result<int32_t> document::get() & noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> document::get() & noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> document::get() & noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> document::get() & noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> document::get() & noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> document::get() & noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> document::get() & noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<> simdjson_inline simdjson_result<bool> document::get() & noexcept { return get_bool(); }
 template<> simdjson_inline simdjson_result<value> document::get() & noexcept { return get_value(); }

@@ -83154,17 +104764,35 @@ template<> simdjson_warn_unused simdjson_inline error_code document::get(array&
 template<> simdjson_warn_unused simdjson_inline error_code document::get(object& out) & noexcept { return get_object().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(raw_json_string& out) & noexcept { return get_raw_json_string().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(std::string_view& out) & noexcept { return get_string(false).get(out); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::u8string_view& out) & noexcept { return get_u8string(false).get(out); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_warn_unused simdjson_inline error_code document::get(double& out) & noexcept { return get_double().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(float& out) & noexcept { return get_float().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(uint64_t& out) & noexcept { return get_uint64().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(int64_t& out) & noexcept { return get_int64().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(uint32_t& out) & noexcept { return get_uint32().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(int32_t& out) & noexcept { return get_int32().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(uint16_t& out) & noexcept { return get_uint16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(int16_t& out) & noexcept { return get_int16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(uint8_t& out) & noexcept { return get_uint8().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(int8_t& out) & noexcept { return get_int8().get(out); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::float32_t& out) & noexcept { return get_float32().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::float64_t& out) & noexcept { return get_float64().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<> simdjson_warn_unused simdjson_inline error_code document::get(bool& out) & noexcept { return get_bool().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(value& out) & noexcept { return get_value().get(out); }

 template<> simdjson_deprecated simdjson_inline simdjson_result<raw_json_string> document::get() && noexcept { return get_raw_json_string(); }
 template<> simdjson_deprecated simdjson_inline simdjson_result<std::string_view> document::get() && noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_deprecated simdjson_inline simdjson_result<std::u8string_view> document::get() && noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_deprecated simdjson_inline simdjson_result<double> document::get() && noexcept { return std::forward<document>(*this).get_double(); }
+template<> simdjson_deprecated simdjson_inline simdjson_result<float> document::get() && noexcept { return std::forward<document>(*this).get_float(); }
 template<> simdjson_deprecated simdjson_inline simdjson_result<uint64_t> document::get() && noexcept { return std::forward<document>(*this).get_uint64(); }
 template<> simdjson_deprecated simdjson_inline simdjson_result<int64_t> document::get() && noexcept { return std::forward<document>(*this).get_int64(); }
 template<> simdjson_deprecated simdjson_inline simdjson_result<bool> document::get() && noexcept { return std::forward<document>(*this).get_bool(); }
@@ -83503,6 +105131,22 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<fallback::ondemand::doc
   if (error()) { return error(); }
   return first.get_int32();
 }
+simdjson_inline simdjson_result<uint16_t> simdjson_result<fallback::ondemand::document>::get_uint16() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<fallback::ondemand::document>::get_int16() noexcept {
+  if (error()) { return error(); }
+  return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<fallback::ondemand::document>::get_uint8() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<fallback::ondemand::document>::get_int8() noexcept {
+  if (error()) { return error(); }
+  return first.get_int8();
+}
 simdjson_inline simdjson_result<double> simdjson_result<fallback::ondemand::document>::get_double() noexcept {
   if (error()) { return error(); }
   return first.get_double();
@@ -83511,10 +105155,36 @@ simdjson_inline simdjson_result<double> simdjson_result<fallback::ondemand::docu
   if (error()) { return error(); }
   return first.get_double_in_string();
 }
+simdjson_inline simdjson_result<float> simdjson_result<fallback::ondemand::document>::get_float() noexcept {
+  if (error()) { return error(); }
+  return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<fallback::ondemand::document>::get_float_in_string() noexcept {
+  if (error()) { return error(); }
+  return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<fallback::ondemand::document>::get_float32() noexcept {
+  if (error()) { return error(); }
+  return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<fallback::ondemand::document>::get_float64() noexcept {
+  if (error()) { return error(); }
+  return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<std::string_view> simdjson_result<fallback::ondemand::document>::get_string(bool allow_replacement) noexcept {
   if (error()) { return error(); }
   return first.get_string(allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<fallback::ondemand::document>::get_u8string(bool allow_replacement) noexcept {
+  if (error()) { return error(); }
+  return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code simdjson_result<fallback::ondemand::document>::get_string(string_type& receiver, bool allow_replacement) noexcept {
   if (error()) { return error(); }
@@ -83542,22 +105212,46 @@ simdjson_inline simdjson_result<bool> simdjson_result<fallback::ondemand::docume
 }

 template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<fallback::ondemand::document>::get() & noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<fallback::ondemand::document>::get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, fallback::ondemand::document>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>();
 }
 template<typename T>
-simdjson_deprecated simdjson_inline simdjson_result<T> simdjson_result<fallback::ondemand::document>::get() && noexcept {
+simdjson_deprecated simdjson_inline simdjson_result<T> simdjson_result<fallback::ondemand::document>::get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, fallback::ondemand::document>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return std::forward<fallback::ondemand::document>(first).get<T>();
 }
 template<typename T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<fallback::ondemand::document>::get(T &out) & noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<fallback::ondemand::document>::get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, fallback::ondemand::document>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>(out);
 }
 template<typename T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<fallback::ondemand::document>::get(T &out) && noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<fallback::ondemand::document>::get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, fallback::ondemand::document>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return std::forward<fallback::ondemand::document>(first).get<T>(out);
 }
@@ -83626,27 +105320,27 @@ simdjson_inline simdjson_result<fallback::ondemand::document>::operator fallback
 }
 simdjson_inline simdjson_result<fallback::ondemand::document>::operator uint64_t() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_uint64();
 }
 simdjson_inline simdjson_result<fallback::ondemand::document>::operator int64_t() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_int64();
 }
 simdjson_inline simdjson_result<fallback::ondemand::document>::operator double() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_double();
 }
 simdjson_inline simdjson_result<fallback::ondemand::document>::operator std::string_view() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_string();
 }
 simdjson_inline simdjson_result<fallback::ondemand::document>::operator fallback::ondemand::raw_json_string() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_raw_json_string();
 }
 simdjson_inline simdjson_result<fallback::ondemand::document>::operator bool() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_bool();
 }
 simdjson_inline simdjson_result<fallback::ondemand::document>::operator fallback::ondemand::value() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
@@ -83736,21 +105430,38 @@ simdjson_inline simdjson_result<uint64_t> document_reference::get_uint64() noexc
 simdjson_inline simdjson_result<uint64_t> document_reference::get_uint64_in_string() noexcept { return doc->get_root_value_iterator().get_root_uint64_in_string(false); }
 simdjson_inline simdjson_result<int64_t> document_reference::get_int64() noexcept { return doc->get_root_value_iterator().get_root_int64(false); }
 simdjson_inline simdjson_result<int64_t> document_reference::get_int64_in_string() noexcept { return doc->get_root_value_iterator().get_root_int64_in_string(false); }
-simdjson_inline simdjson_result<uint32_t> document_reference::get_uint32() noexcept {
-  uint64_t result;
-  SIMDJSON_TRY(doc->get_root_value_iterator().get_root_uint64(false).get(result));
-  if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<uint32_t>(result);
-}
-simdjson_inline simdjson_result<int32_t> document_reference::get_int32() noexcept {
-  int64_t result;
-  SIMDJSON_TRY(doc->get_root_value_iterator().get_root_int64(false).get(result));
-  if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<int32_t>(result);
-}
+simdjson_inline simdjson_result<uint32_t> document_reference::get_uint32() noexcept { return narrow_integer<uint32_t>(get_uint64()); }
+simdjson_inline simdjson_result<int32_t> document_reference::get_int32() noexcept { return narrow_integer<int32_t>(get_int64()); }
+simdjson_inline simdjson_result<uint16_t> document_reference::get_uint16() noexcept { return narrow_integer<uint16_t>(get_uint64()); }
+simdjson_inline simdjson_result<int16_t> document_reference::get_int16() noexcept { return narrow_integer<int16_t>(get_int64()); }
+simdjson_inline simdjson_result<uint8_t> document_reference::get_uint8() noexcept { return narrow_integer<uint8_t>(get_uint64()); }
+simdjson_inline simdjson_result<int8_t> document_reference::get_int8() noexcept { return narrow_integer<int8_t>(get_int64()); }
 simdjson_inline simdjson_result<double> document_reference::get_double() noexcept { return doc->get_root_value_iterator().get_root_double(false); }
-simdjson_inline simdjson_result<double> document_reference::get_double_in_string() noexcept { return doc->get_root_value_iterator().get_root_double(false); }
+simdjson_inline simdjson_result<double> document_reference::get_double_in_string() noexcept { return doc->get_root_value_iterator().get_root_double_in_string(false); }
+simdjson_inline simdjson_result<float> document_reference::get_float() noexcept { return doc->get_root_value_iterator().get_root_float(false); }
+simdjson_inline simdjson_result<float> document_reference::get_float_in_string() noexcept { return doc->get_root_value_iterator().get_root_float_in_string(false); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> document_reference::get_float32() noexcept {
+  float result;
+  SIMDJSON_TRY(get_float().get(result));
+  return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> document_reference::get_float64() noexcept {
+  double result;
+  SIMDJSON_TRY(get_double().get(result));
+  return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<std::string_view> document_reference::get_string(bool allow_replacement) noexcept { return doc->get_root_value_iterator().get_root_string(false, allow_replacement); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> document_reference::get_u8string(bool allow_replacement) noexcept {
+  std::string_view content;
+  SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+  return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code document_reference::get_string(string_type& receiver, bool allow_replacement) noexcept { return doc->get_root_value_iterator().get_root_string(receiver, false, allow_replacement); }
 simdjson_inline simdjson_result<std::string_view> document_reference::get_wobbly_string() noexcept { return doc->get_root_value_iterator().get_root_wobbly_string(false); }
@@ -83762,11 +105473,25 @@ template<> simdjson_inline simdjson_result<array> document_reference::get() & no
 template<> simdjson_inline simdjson_result<object> document_reference::get() & noexcept { return get_object(); }
 template<> simdjson_inline simdjson_result<raw_json_string> document_reference::get() & noexcept { return get_raw_json_string(); }
 template<> simdjson_inline simdjson_result<std::string_view> document_reference::get() & noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> document_reference::get() & noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_inline simdjson_result<double> document_reference::get() & noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> document_reference::get() & noexcept { return get_float(); }
 template<> simdjson_inline simdjson_result<uint64_t> document_reference::get() & noexcept { return get_uint64(); }
 template<> simdjson_inline simdjson_result<int64_t> document_reference::get() & noexcept { return get_int64(); }
 template<> simdjson_inline simdjson_result<uint32_t> document_reference::get() & noexcept { return get_uint32(); }
 template<> simdjson_inline simdjson_result<int32_t> document_reference::get() & noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> document_reference::get() & noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> document_reference::get() & noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> document_reference::get() & noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> document_reference::get() & noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> document_reference::get() & noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> document_reference::get() & noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<> simdjson_inline simdjson_result<bool> document_reference::get() & noexcept { return get_bool(); }
 template<> simdjson_inline simdjson_result<value> document_reference::get() & noexcept { return get_value(); }
 #if SIMDJSON_EXCEPTIONS
@@ -83912,6 +105637,22 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<fallback::ondemand::doc
   if (error()) { return error(); }
   return first.get_int32();
 }
+simdjson_inline simdjson_result<uint16_t> simdjson_result<fallback::ondemand::document_reference>::get_uint16() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<fallback::ondemand::document_reference>::get_int16() noexcept {
+  if (error()) { return error(); }
+  return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<fallback::ondemand::document_reference>::get_uint8() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<fallback::ondemand::document_reference>::get_int8() noexcept {
+  if (error()) { return error(); }
+  return first.get_int8();
+}
 simdjson_inline simdjson_result<double> simdjson_result<fallback::ondemand::document_reference>::get_double() noexcept {
   if (error()) { return error(); }
   return first.get_double();
@@ -83920,10 +105661,36 @@ simdjson_inline simdjson_result<double> simdjson_result<fallback::ondemand::docu
   if (error()) { return error(); }
   return first.get_double_in_string();
 }
+simdjson_inline simdjson_result<float> simdjson_result<fallback::ondemand::document_reference>::get_float() noexcept {
+  if (error()) { return error(); }
+  return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<fallback::ondemand::document_reference>::get_float_in_string() noexcept {
+  if (error()) { return error(); }
+  return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<fallback::ondemand::document_reference>::get_float32() noexcept {
+  if (error()) { return error(); }
+  return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<fallback::ondemand::document_reference>::get_float64() noexcept {
+  if (error()) { return error(); }
+  return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<std::string_view> simdjson_result<fallback::ondemand::document_reference>::get_string(bool allow_replacement) noexcept {
   if (error()) { return error(); }
   return first.get_string(allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<fallback::ondemand::document_reference>::get_u8string(bool allow_replacement) noexcept {
+  if (error()) { return error(); }
+  return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code simdjson_result<fallback::ondemand::document_reference>::get_string(string_type& receiver, bool allow_replacement) noexcept {
   if (error()) { return error(); }
@@ -83950,22 +105717,46 @@ simdjson_inline simdjson_result<bool> simdjson_result<fallback::ondemand::docume
   return first.is_null();
 }
 template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<fallback::ondemand::document_reference>::get() & noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<fallback::ondemand::document_reference>::get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, fallback::ondemand::document_reference>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>();
 }
 template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<fallback::ondemand::document_reference>::get() && noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<fallback::ondemand::document_reference>::get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, fallback::ondemand::document_reference>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return std::forward<fallback::ondemand::document_reference>(first).get<T>();
 }
 template <class T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<fallback::ondemand::document_reference>::get(T &out) & noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<fallback::ondemand::document_reference>::get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, fallback::ondemand::document_reference>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>(out);
 }
 template <class T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<fallback::ondemand::document_reference>::get(T &out) && noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<fallback::ondemand::document_reference>::get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, fallback::ondemand::document_reference>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return std::forward<fallback::ondemand::document_reference>(first).get<T>(out);
 }
@@ -84027,27 +105818,27 @@ simdjson_inline simdjson_result<fallback::ondemand::document_reference>::operato
 }
 simdjson_inline simdjson_result<fallback::ondemand::document_reference>::operator uint64_t() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_uint64();
 }
 simdjson_inline simdjson_result<fallback::ondemand::document_reference>::operator int64_t() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_int64();
 }
 simdjson_inline simdjson_result<fallback::ondemand::document_reference>::operator double() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_double();
 }
 simdjson_inline simdjson_result<fallback::ondemand::document_reference>::operator std::string_view() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_string();
 }
 simdjson_inline simdjson_result<fallback::ondemand::document_reference>::operator fallback::ondemand::raw_json_string() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_raw_json_string();
 }
 simdjson_inline simdjson_result<fallback::ondemand::document_reference>::operator bool() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_bool();
 }
 simdjson_inline simdjson_result<fallback::ondemand::document_reference>::operator fallback::ondemand::value() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
@@ -84113,6 +105904,7 @@ simdjson_warn_unused simdjson_inline error_code simdjson_result<fallback::ondema
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */

 #include <algorithm>
+#include <cstring>
 #include <stdexcept>

 namespace simdjson {
@@ -84199,23 +105991,20 @@ simdjson_inline document_stream::document_stream(
   const uint8_t *_buf,
   size_t _len,
   size_t _batch_size,
-  bool _allow_comma_separated
+  bool _allow_comma_separated,
+  stream_format _format
 ) noexcept
   : parser{&_parser},
     buf{_buf},
     len{_len},
     batch_size{_batch_size <= MINIMAL_BATCH_SIZE ? MINIMAL_BATCH_SIZE : _batch_size},
     allow_comma_separated{_allow_comma_separated},
+    format{_format},
     error{SUCCESS}
     #ifdef SIMDJSON_THREADS_ENABLED
     , use_thread(_parser.threaded) // we need to make a copy because _parser.threaded can change
     #endif
 {
-#ifdef SIMDJSON_THREADS_ENABLED
-  if(worker.get() == nullptr) {
-    error = MEMALLOC;
-  }
-#endif
 }

 simdjson_inline document_stream::document_stream() noexcept
@@ -84224,6 +106013,7 @@ simdjson_inline document_stream::document_stream() noexcept
     len{0},
     batch_size{0},
     allow_comma_separated{false},
+    format{stream_format::whitespace_delimited},
     error{UNINITIALIZED}
     #ifdef SIMDJSON_THREADS_ENABLED
     , use_thread(false)
@@ -84243,6 +106033,9 @@ inline size_t document_stream::size_in_bytes() const noexcept {
 }

 inline size_t document_stream::truncated_bytes() const noexcept {
+  // Stage 1 returns EMPTY on zero-length input before it writes the index
+  // sentinels read below, so they would still hold a previous stream's values.
+  if (len == 0) { return 0; }
   if(error == CAPACITY) { return len - batch_start; }
   return parser->implementation->structural_indexes[parser->implementation->n_structural_indexes] - parser->implementation->structural_indexes[parser->implementation->n_structural_indexes + 1];
 }
@@ -84323,13 +106116,20 @@ inline void document_stream::start() noexcept {
     error = run_stage1(*parser, batch_start);
   }
   if (error) { return; }
-  doc_index = batch_start;
+  // For json_sequence mode, structural_indexes[0] points to the actual JSON value
+  // after the RS delimiter and any following whitespace. For regular mode, it is
+  // the offset from batch_start to the first document in the batch.
+  doc_index = batch_start + parser->implementation->structural_indexes[0];
   doc = document(json_iterator(&buf[batch_start], parser));
   doc.iter._streaming = true;

   #ifdef SIMDJSON_THREADS_ENABLED
   if (use_thread && next_batch_start() < len) {
     // Kick off the first thread on next batch if needed
+    if (worker.get() == nullptr) {
+      worker.reset(new(std::nothrow) stage1_worker());
+      if (worker.get() == nullptr) { error = MEMALLOC; return; }
+    }
     error = stage1_thread_parser.allocate(batch_size);
     if (error) { return; }
     worker->start_thread();
@@ -84404,12 +106204,69 @@ inline void document_stream::next() noexcept {
        */

       if (error) { continue; } // If the error was EMPTY, we may want to load another batch.
-      doc_index = batch_start;
+      doc_index = batch_start + parser->implementation->structural_indexes[0];
     }
   }
 }

+simdjson_inline uint8_t document_stream::document_delimiter() const noexcept {
+  switch (format) {
+    case stream_format::newline_delimited: return '\n';
+    case stream_format::json_sequence: return 0x1E;
+    default: return 0;
+  }
+}
+
+simdjson_inline bool document_stream::skip_to_delimiter(uint8_t delimiter) noexcept {
+  const uint8_t *const base = &buf[batch_start];
+  const token_position pos = doc.iter.position();
+  const token_position end = doc.iter.end_position();
+  if (pos >= end) { return false; }
+  const size_t here = size_t(doc.iter.token.peek(pos) - base);
+  const size_t batch_len =
+      (len - batch_start < batch_size) ? len - batch_start : batch_size;
+  if (here >= batch_len) { return false; }
+  const uint8_t *const found = static_cast<const uint8_t *>(
+      std::memchr(base + here, delimiter, batch_len - here));
+  if (found == nullptr) { return false; }
+
+  const uint32_t boundary = uint32_t(found - base);
+  // The answer is near `pos`: the delimiter ends the current document, while
+  // `end` spans the whole batch. Gallop first so the cost follows the distance
+  // rather than the size of the batch.
+  token_position lo = pos;
+  size_t hop = 1;
+  while (lo + hop < end && lo[hop] < boundary) { lo += hop; hop <<= 1; }
+  token_position hi = (lo + hop < end) ? lo + hop : end;
+  while (lo < hi) {
+    const token_position mid = lo + ((hi - lo) >> 1);
+    if (*mid < boundary) { lo = mid + 1; } else { hi = mid; }
+  }
+  doc.iter.token.set_position(lo);
+  return true;
+}
+
 inline void document_stream::next_document() noexcept {
+  // A delimiter that cannot occur inside a document tells us where the current
+  // one ends, so we can jump there instead of walking every structural. Only
+  // valid while the iterator is still inside the document: a consumed document
+  // already sits on the next one's first token, and skip_child() returns at
+  // once for it.
+  //
+  // The jump does not structure-validate the unread remainder of the current
+  // document: under newline_delimited / json_sequence the next delimiter is
+  // assumed to be the true document boundary. Callers that leave depth() > 0
+  // while violating that contract (e.g. pretty multi-line JSON under
+  // newline_delimited) can mis-align following documents; use
+  // whitespace_delimited if unsure.
+  const uint8_t delimiter = document_delimiter();
+  if (delimiter != 0 && !error && doc.iter.depth() > 0 &&
+      skip_to_delimiter(delimiter)) {
+    doc.iter._depth = 1;
+    doc.iter._string_buf_loc = parser->string_buf.get();
+    doc.iter._root = doc.iter.position();
+    return;
+  }
   // Go to next place where depth=0 (document depth)
   error = doc.iter.skip_child(0);
   if (error) { return; }
@@ -84433,10 +106290,35 @@ inline error_code document_stream::run_stage1(ondemand::parser &p, size_t _batch
   // This code only updates the structural index in the parser, it does not update any json_iterator
   // instance.
   size_t remaining = len - _batch_start;
+  stage1_mode mode;
   if (remaining <= batch_size) {
-    return p.implementation->stage1(&buf[_batch_start], remaining, stage1_mode::streaming_final);
+    // Final batch
+    switch (format) {
+      case stream_format::json_sequence:
+        mode = stage1_mode::json_sequence_final;
+        break;
+      case stream_format::comma_delimited:
+        mode = stage1_mode::comma_delimited_final;
+        break;
+      default:
+        mode = stage1_mode::streaming_final;
+        break;
+    }
+    return p.implementation->stage1(&buf[_batch_start], remaining, mode);
   } else {
-    return p.implementation->stage1(&buf[_batch_start], batch_size, stage1_mode::streaming_partial);
+    // Partial batch
+    switch (format) {
+      case stream_format::json_sequence:
+        mode = stage1_mode::json_sequence_partial;
+        break;
+      case stream_format::comma_delimited:
+        mode = stage1_mode::comma_delimited_partial;
+        break;
+      default:
+        mode = stage1_mode::streaming_partial;
+        break;
+    }
+    return p.implementation->stage1(&buf[_batch_start], batch_size, mode);
   }
 }

@@ -84445,11 +106327,19 @@ simdjson_inline size_t document_stream::iterator::current_index() const noexcept
 }

 simdjson_inline std::string_view document_stream::iterator::source() const noexcept {
-  auto depth = stream->doc.iter.depth();
+  // On error (e.g., CAPACITY), there is no document to walk: return the rest of
+  // the input, as the DOM document_stream does.
+  if (stream->error) {
+    return std::string_view(reinterpret_cast<const char*>(stream->buf) + current_index(), stream->len - current_index());
+  }
+  // Always walk from the root of the document, whatever the current position
+  // of the document iterator: the user may have already consumed part of the
+  // document, so the iterator's current depth must not be used here.
+  depth_t depth = 1;
   auto cur_struct_index = stream->doc.iter._root - stream->parser->implementation->structural_indexes.get();

-  // If at root, process the first token to determine if scalar value
-  if (stream->doc.iter.at_root()) {
+  // Process the first token to determine if scalar value
+  {
     switch (stream->buf[stream->batch_start + stream->parser->implementation->structural_indexes[cur_struct_index]]) {
       case '{': case '[':   // Depth=1 already at start of document
         break;
@@ -84457,14 +106347,47 @@ simdjson_inline std::string_view document_stream::iterator::source() const noexc
         depth--;
         break;
       default:    // Scalar value document
-        // TODO: We could remove trailing whitespaces
         // This returns a string spanning from start of value to the beginning of the next document (excluded)
         {
           auto next_index = stream->batch_start + stream->parser->implementation->structural_indexes[++cur_struct_index];
           // normally the length would be next_index - current_index() - 1, except for the last document
           size_t svlen = next_index - current_index();
           const char *start = reinterpret_cast<const char*>(stream->buf) + current_index();
-          while(svlen > 1 && (std::isspace(start[svlen-1]) || start[svlen-1] == '\0')) {
+          // When the scalar is followed by a truncated document, the structural
+          // indexes of that document were dropped and next_index is the end of
+          // the input, so we bound the scalar by scanning the token itself.
+          size_t token_len = 0;
+          if (*start == '"') {
+            token_len = 1;
+            while (token_len < svlen) {
+              char c = start[token_len++];
+              if (c == '\\') {
+                token_len++;
+              } else if (c == '"') {
+                break;
+              }
+            }
+          } else {
+            while (token_len < svlen) {
+              char c = start[token_len];
+              if (std::isspace(static_cast<unsigned char>(c)) || c == ',' || c == '{' || c == '[' || c == '\0' || static_cast<uint8_t>(c) == 0x1E) {
+                break;
+              }
+              token_len++;
+            }
+          }
+          if (token_len > 0 && token_len < svlen) {
+            svlen = token_len;
+          }
+          // Trim trailing whitespace, NUL, and RS (0x1E). In RFC 7464
+          // json_sequence mode the scanner classifies RS as a scalar
+          // character, so an RS-prefixed scalar document (number / true /
+          // false / null / string) has no closing structural index and the
+          // slice runs all the way up to the next document's RS. RS cannot
+          // legally appear in a JSON value at the source level (control
+          // characters in strings must be escaped as \u001E), so stripping
+          // it is safe in every stream_format.
+          while(svlen > 1 && (std::isspace(static_cast<unsigned char>(start[svlen-1])) || start[svlen-1] == '\0' || static_cast<uint8_t>(start[svlen-1]) == 0x1E || (stream->format == stream_format::comma_delimited && start[svlen-1] == ','))) {
             svlen--;
           }
           return std::string_view(start, svlen);
@@ -84589,11 +106512,19 @@ simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> field::un
   return answer;
 }

+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_warn_unused simdjson_result<std::u8string_view> field::unescaped_u8key(bool allow_replacement) noexcept {
+  std::string_view key;
+  SIMDJSON_TRY( unescaped_key(allow_replacement).get(key) );
+  return internal::as_u8string_view(key);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 template <typename string_type>
 simdjson_inline simdjson_warn_unused error_code field::unescaped_key(string_type& receiver, bool allow_replacement) noexcept {
   std::string_view key;
   SIMDJSON_TRY( unescaped_key(allow_replacement).get(key) );
-  receiver = key;
+  internal::assign_utf8(receiver, key);
   return SUCCESS;
 }

@@ -84615,6 +106546,12 @@ simdjson_inline std::string_view field::escaped_key() const noexcept {
   return std::string_view(reinterpret_cast<const char*>(first.buf), end_quote - first.buf);
 }

+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline std::u8string_view field::escaped_u8key() const noexcept {
+  return internal::as_u8string_view(escaped_key());
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 simdjson_inline value &field::value() & noexcept {
   return second;
 }
@@ -84659,11 +106596,25 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<fallback::onde
   return first.escaped_key();
 }

+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<fallback::ondemand::field>::escaped_u8key() noexcept {
+  if (error()) { return error(); }
+  return first.escaped_u8key();
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 simdjson_inline simdjson_result<std::string_view> simdjson_result<fallback::ondemand::field>::unescaped_key(bool allow_replacement) noexcept {
   if (error()) { return error(); }
   return first.unescaped_key(allow_replacement);
 }

+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<fallback::ondemand::field>::unescaped_u8key(bool allow_replacement) noexcept {
+  if (error()) { return error(); }
+  return first.unescaped_u8key(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 template<typename string_type>
 simdjson_warn_unused simdjson_inline error_code simdjson_result<fallback::ondemand::field>::unescaped_key(string_type &receiver, bool allow_replacement) noexcept {
   if (error()) { return error(); }
@@ -84707,6 +106658,9 @@ simdjson_inline json_iterator::json_iterator(json_iterator &&other) noexcept
     _depth{other._depth},
     _root{other._root},
     _streaming{other._streaming}
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+    , _allow_incomplete_json{other._allow_incomplete_json}
+#endif
 {
   other.parser = nullptr;
 }
@@ -84718,6 +106672,9 @@ simdjson_inline json_iterator &json_iterator::operator=(json_iterator &&other) n
   _depth = other._depth;
   _root = other._root;
   _streaming = other._streaming;
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  _allow_incomplete_json = other._allow_incomplete_json;
+#endif
   other.parser = nullptr;
   return *this;
 }
@@ -84744,7 +106701,8 @@ simdjson_inline json_iterator::json_iterator(const uint8_t *buf, ondemand::parse
       _string_buf_loc{parser->string_buf.get()},
       _depth{1},
       _root{parser->implementation->structural_indexes.get()},
-      _streaming{streaming}
+      _streaming{streaming},
+      _allow_incomplete_json{true}

 {
   logger::log_headers();
@@ -84816,7 +106774,8 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::skip_child(depth_
 #endif // SIMDJSON_CHECK_EOF
       break;
     case '"':
-      if(*peek() == ':') {
+      // At the end, peek() would read the sentinel, which points into the padding.
+      if(!at_end() && *peek() == ':') {
         // We are at a key!!!
         // This might happen if you just started an object and you skip it immediately.
         // Performance note: it would be nice to get rid of this check as it is somewhat
@@ -84859,7 +106818,7 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::skip_child(depth_
     }
   }

-  return report_error(TAPE_ERROR, "not enough close braces");
+  return report_error(INCOMPLETE_ARRAY_OR_OBJECT, "not enough close braces");
 }

 SIMDJSON_POP_DISABLE_WARNINGS
@@ -84876,6 +106835,17 @@ simdjson_inline bool json_iterator::streaming() const noexcept {
   return _streaming;
 }

+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+simdjson_inline bool json_iterator::allow_incomplete_json() const noexcept {
+  return _allow_incomplete_json;
+}
+
+simdjson_inline size_t json_iterator::remaining_input_length(const uint8_t *json) const noexcept {
+  const uint8_t *end = token.buf + parser->_document_len;
+  return json < end ? size_t(end - json) : 0;
+}
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+
 simdjson_inline token_position json_iterator::root_position() const noexcept {
   return _root;
 }
@@ -85158,7 +107128,7 @@ inline std::ostream& operator<<(std::ostream& out, json_type type) noexcept {
         case json_type::string: out << "string"; break;
         case json_type::boolean: out << "boolean"; break;
         case json_type::null: out << "null"; break;
-        default: SIMDJSON_UNREACHABLE();
+        case json_type::unknown: out << "unknown"; break;
     }
     return out;
 }
@@ -85497,6 +107467,10 @@ inline void log_line(const json_iterator &iter, token_position index, depth_t de
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_iterator.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value-inl.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/jsonpathutil.h" */
+/* amalgamation skipped (editor-only): #include <utility> */
+/* amalgamation skipped (editor-only): #if SIMDJSON_SUPPORTS_CONCEPTS */
+/* amalgamation skipped (editor-only): #include <tuple> // std::forward_as_tuple/get for the variadic for_each adapter */
+/* amalgamation skipped (editor-only): #endif */
 /* amalgamation skipped (editor-only): #if SIMDJSON_STATIC_REFLECTION */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_string_builder.h"  // for constevalutil::fixed_string */
 /* amalgamation skipped (editor-only): #include <meta> */
@@ -85526,12 +107500,21 @@ simdjson_inline simdjson_result<value> object::find_field_unordered(const std::s
   return value(iter.child());
 }
 simdjson_inline simdjson_result<value> object::operator[](const std::string_view key) & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   return find_field_unordered(key);
 }
 simdjson_inline simdjson_result<value> object::operator[](const std::string_view key) && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   return std::forward<object>(*this).find_field_unordered(key);
 }
 simdjson_inline simdjson_result<value> object::find_field(const std::string_view key) & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   bool has_value;
   SIMDJSON_TRY( iter.find_field_raw(key).get(has_value) );
   if (!has_value) {
@@ -85541,6 +107524,9 @@ simdjson_inline simdjson_result<value> object::find_field(const std::string_view
   return value(iter.child());
 }
 simdjson_inline simdjson_result<value> object::find_field(const std::string_view key) && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   bool has_value;
   SIMDJSON_TRY( iter.find_field_raw(key).get(has_value) );
   if (!has_value) {
@@ -85550,6 +107536,150 @@ simdjson_inline simdjson_result<value> object::find_field(const std::string_view
   return value(iter.child());
 }

+#if SIMDJSON_SUPPORTS_CONCEPTS
+template <typename Selector, typename Func>
+  requires key_selector_type<Selector> &&
+           std::is_invocable_v<Func&, std::size_t, value>
+simdjson_flatten simdjson_inline for_each_result object::for_each(Func&& on_match)
+    noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, value>) {
+  // Single pass driven directly by the value_iterator, mirroring
+  // find_field_unordered_raw + value(iter.child()). Compared to walking via
+  // object_iterator/field, this avoids constructing a simdjson_result<field> and
+  // a field (key + value) for every field -- and the development-check bookkeeping
+  // in object_iterator -- building a value only for the fields that actually match.
+  // We operate on a copy of the iterator, as object::begin() would.
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  // Mirror object::begin(): for_each must start at the beginning of the object,
+  // not from some position left behind by a prior find_field on the same object.
+  if (!iter.is_at_iterator_start()) { return {OUT_OF_ORDER_ITERATION, 0}; }
+#endif
+  value_iterator it = iter;
+  std::size_t matched = 0;
+  // Track which selector indices have already matched, as a compile-time bitset
+  // (one 64-bit word per 64 keys): the callback fires once per key, on its first
+  // occurrence, and we stop as soon as every key has matched.
+  constexpr std::size_t seen_words = (Selector::size() + 63) / 64;
+  std::array<std::uint64_t, seen_words> seen{};
+  while (it.is_open()) {
+    raw_json_string key;
+    error_code error;
+    std::size_t idx;
+    if constexpr (Selector::window.ok) {
+      // A window selector confirms a key from its raw bytes alone (the closing
+      // quote bounds it), so we take the length-free path: field_key (no backward
+      // scan, mirroring the ordered find_field) feeding match_raw(raw_json_string).
+      if ((error = it.field_key().get(key))) { it.abandon(); return {error, matched}; }
+      if ((error = it.field_value())) { it.abandon(); return {error, matched}; }
+      idx = Selector::match_raw(key);
+    } else {
+      // Otherwise derive the key length from the structural index (the following
+      // ':' token) to feed the length-aware hash, avoiding a forward SIMD scan.
+      std::size_t key_len;
+      if ((error = it.field_key_with_length(key, key_len))) { it.abandon(); return {error, matched}; }
+      if ((error = it.field_value())) { it.abandon(); return {error, matched}; }
+      idx = Selector::match_raw(key.raw(), key_len);
+    }
+    if (idx < Selector::size()) {
+      const std::uint64_t seen_bit = std::uint64_t{1} << (idx & 63);
+      std::uint64_t &seen_word = seen[idx >> 6];
+      if (!(seen_word & seen_bit)) {
+        seen_word |= seen_bit;
+        value matched_value(it.child());
+        // The callback may return void or anything convertible to error_code
+        // (error_code itself, or a for_each_result from a nested for_each). When
+        // it yields an error_code, we stop at the first non-SUCCESS result and
+        // propagate it so the caller can surface value-parse errors (for example,
+        // a type mismatch on a matched field). A void-returning callback is
+        // responsible for handling its own errors.
+        if constexpr (std::is_convertible_v<decltype(on_match(idx, matched_value)), error_code>) {
+          // Unlike the internal-error paths above, a callback error does not
+          // abandon the iterator: we leave it recoverable so the caller can keep
+          // using the object (or its parent) after handling the error.
+          if ((error = on_match(idx, matched_value))) { return {error, matched}; }
+        } else {
+          on_match(idx, matched_value);
+        }
+        if (++matched >= Selector::size()) { break; }
+      }
+    }
+    // Mirror object_iterator::operator++'s safety rail: if the callback consumed
+    // the value and left the iterator closed or in error (e.g. a void callback
+    // that swallowed a fatal sub-iteration error), stop here rather than calling
+    // skip_child on a closed iterator.
+    if (!it.is_open()) { break; }
+    // Skip the value (a no-op if the callback consumed it) and step to the next
+    // field; has_next_field() ends the container on '}', which closes the loop.
+    if ((error = it.skip_child())) { it.abandon(); return {error, matched}; }
+    if ((error = it.has_next_field().error())) { return {error, matched}; }
+  }
+  return {SUCCESS, matched};
+}
+
+namespace key_selector_for_each_detail {
+
+// Dispatch a matched value to the I-th handler in the tuple (0-based). A handler
+// is either an invocable taking the value (void- or error_code-returning) or a
+// deserialization target, in which case we assign via value::get. We always
+// return error_code so the core (index, value) for_each can treat the adapter
+// uniformly.
+template <typename Tuple, std::size_t... Is>
+simdjson_really_inline error_code dispatch_value(
+    std::size_t idx, Tuple& handlers, value v, std::index_sequence<Is...>) {
+  error_code err = SUCCESS;
+  auto try_one = [&](auto Ic) {
+    constexpr std::size_t I = decltype(Ic)::value;
+    if (idx == I) {
+      auto&& h = std::get<I>(handlers);
+      using H = std::remove_reference_t<decltype(h)>;
+      if constexpr (std::is_invocable_v<H&, value>) {
+        // A handler returning void runs for its side effects; one returning
+        // anything convertible to error_code (error_code, or a for_each_result
+        // from a nested for_each) has its error captured and propagated.
+        if constexpr (std::is_convertible_v<decltype(h(v)), error_code>) {
+          err = h(v);
+        } else {
+          h(v);
+        }
+      } else {
+        // Direct deserialization target: assign the matched value into it.
+        err = v.get(h);
+      }
+    }
+  };
+  (try_one(std::integral_constant<std::size_t, Is>{}), ...);
+  return err;
+}
+
+} // namespace key_selector_for_each_detail
+
+template <typename Selector, typename... Handlers>
+  requires key_selector_type<Selector> &&
+           (sizeof...(Handlers) == Selector::size()) &&
+           (key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_flatten simdjson_inline for_each_result object::for_each(Handlers&&... on_match)
+    noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+  auto handlers = std::forward_as_tuple(std::forward<Handlers>(on_match)...);
+  // Reuse the single (index, value) implementation via a tiny adapter.
+  // The adapter is called once per *matched* key (very few); the hot path
+  // (iteration + match_raw + seen bitset) stays exactly the same.
+  return this->template for_each<Selector>(
+      [&](std::size_t i, value v) -> error_code {
+        return key_selector_for_each_detail::dispatch_value(
+            i, handlers, v, std::make_index_sequence<Selector::size()>{});
+      });
+}
+
+template <constevalutil::fixed_string... Keys, typename... Handlers>
+  requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+           (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+           (key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline for_each_result object::for_each(Handlers&&... on_match)
+    noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+  using Selector = key_selector<Keys...>;
+  return this->template for_each<Selector>(std::forward<Handlers>(on_match)...);
+}
+#endif
+
 simdjson_inline simdjson_result<object> object::start(value_iterator &iter) noexcept {
   SIMDJSON_TRY( iter.start_object().error() );
   return object(iter);
@@ -85585,6 +107715,9 @@ simdjson_warn_unused simdjson_inline error_code object::consume() noexcept {
 }

 simdjson_inline simdjson_result<std::string_view> object::raw_json() noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   const uint8_t * starting_point{iter.peek_start()};
   auto error = consume();
   if(error) { return error; }
@@ -85606,9 +107739,17 @@ simdjson_inline object::object(const value_iterator &_iter) noexcept
 {
 }

-simdjson_inline simdjson_result<object_iterator> object::begin() noexcept {
+simdjson_inline simdjson_result<object_iterator> object::begin() & noexcept {
 #if SIMDJSON_DEVELOPMENT_CHECKS
-  if (!iter.is_at_iterator_start()) { return OUT_OF_ORDER_ITERATION; }
+  if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
+  return object_iterator(iter, this);
+#endif
+  return object_iterator(iter);
+}
+simdjson_inline simdjson_result<object_iterator> object::begin() && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  // The object is a temporary that the iterator may outlive: do not lock it.
+  if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
 #endif
   return object_iterator(iter);
 }
@@ -85617,7 +107758,9 @@ simdjson_inline simdjson_result<object_iterator> object::end() noexcept {
 }

 inline simdjson_result<value> object::at_pointer(std::string_view json_pointer) noexcept {
-  if (json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
+  // An empty pointer has no json_pointer[0]: with a default-constructed
+  // std::string_view, reading it dereferences a null pointer.
+  if (json_pointer.empty() || json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
   json_pointer = json_pointer.substr(1);
   size_t slash = json_pointer.find('/');
   std::string_view key = json_pointer.substr(0, slash);
@@ -85719,6 +107862,9 @@ simdjson_inline simdjson_result<size_t> object::count_fields() & noexcept {
 }

 simdjson_inline simdjson_result<bool> object::is_empty() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   bool is_not_empty;
   auto error = iter.reset_object().get(is_not_empty);
   if(error) { return error; }
@@ -85726,9 +107872,41 @@ simdjson_inline simdjson_result<bool> object::is_empty() & noexcept {
 }

 simdjson_inline simdjson_result<bool> object::reset() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   return iter.reset_object();
 }

+simdjson_inline object_position object::get_current_position() const noexcept {
+  return object_position{iter.position(), iter.json_iter().depth()};
+}
+
+simdjson_inline error_code object::revert_position(object_position position) noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
+  // json_iterator::reenter_child() requires the live depth to be exactly
+  // one level shallower than the target (matching how every other depth
+  // transition in this iterator works), and under SIMDJSON_DEVELOPMENT_CHECKS
+  // additionally validates against the parser's per-depth container-start
+  // bookkeeping. Neither applies here: depending on what was captured and
+  // what has happened since (a scalar field fully consumed, a compound
+  // value left open, a find_field() miss that scanned past everything),
+  // the live depth when reverting can be any number of levels away from
+  // the captured one, and the captured depth is not necessarily a
+  // container's own start. reenter_at() moves directly, matching how
+  // reset_object() itself repositions without going through reenter_child().
+  iter.reenter_at(position.position, position.depth);
+  return SUCCESS;
+}
+
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline void object::set_locked(bool _locked) noexcept {
+  locked = _locked;
+}
+#endif
+
 #if SIMDJSON_SUPPORTS_CONCEPTS && SIMDJSON_STATIC_REFLECTION

 template<constevalutil::fixed_string... FieldNames, typename T>
@@ -85786,10 +107964,14 @@ simdjson_inline simdjson_result<fallback::ondemand::object>::simdjson_result(fal
 simdjson_inline simdjson_result<fallback::ondemand::object>::simdjson_result(error_code error) noexcept
     : implementation_simdjson_result_base<fallback::ondemand::object>(error) {}

-simdjson_inline simdjson_result<fallback::ondemand::object_iterator> simdjson_result<fallback::ondemand::object>::begin() noexcept {
+simdjson_inline simdjson_result<fallback::ondemand::object_iterator> simdjson_result<fallback::ondemand::object>::begin() & noexcept {
   if (error()) { return error(); }
   return first.begin();
 }
+simdjson_inline simdjson_result<fallback::ondemand::object_iterator> simdjson_result<fallback::ondemand::object>::begin() && noexcept {
+  if (error()) { return error(); }
+  return std::move(first).begin();
+}
 simdjson_inline simdjson_result<fallback::ondemand::object_iterator> simdjson_result<fallback::ondemand::object>::end() noexcept {
   if (error()) { return error(); }
   return first.end();
@@ -85843,11 +108025,55 @@ simdjson_inline error_code simdjson_result<fallback::ondemand::object>::for_each
   return first.for_each_at_path_with_wildcard(json_path, std::forward<Func>(callback));
 }

+#if SIMDJSON_SUPPORTS_CONCEPTS
+template <typename Selector, typename Func>
+  requires fallback::ondemand::key_selector_type<Selector> &&
+           std::is_invocable_v<Func&, std::size_t, fallback::ondemand::value>
+simdjson_inline fallback::ondemand::for_each_result
+simdjson_result<fallback::ondemand::object>::for_each(Func&& on_match)
+    noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, fallback::ondemand::value>) {
+  if (error()) { return {error(), 0}; }
+  return first.template for_each<Selector>(std::forward<Func>(on_match));
+}
+
+template <typename Selector, typename... Handlers>
+  requires fallback::ondemand::key_selector_type<Selector> &&
+           (sizeof...(Handlers) == Selector::size()) &&
+           (fallback::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline fallback::ondemand::for_each_result
+simdjson_result<fallback::ondemand::object>::for_each(Handlers&&... on_match)
+    noexcept(fallback::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+  if (error()) { return {error(), 0}; }
+  return first.template for_each<Selector>(std::forward<Handlers>(on_match)...);
+}
+
+template <constevalutil::fixed_string... Keys, typename... Handlers>
+  requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+           (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+           (fallback::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline fallback::ondemand::for_each_result
+simdjson_result<fallback::ondemand::object>::for_each(Handlers&&... on_match)
+    noexcept(fallback::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+  if (error()) { return {error(), 0}; }
+  return first.template for_each<Keys...>(std::forward<Handlers>(on_match)...);
+}
+#endif
+
 inline simdjson_result<bool> simdjson_result<fallback::ondemand::object>::reset() noexcept {
   if (error()) { return error(); }
   return first.reset();
 }

+inline simdjson_result<fallback::ondemand::object_position> simdjson_result<fallback::ondemand::object>::get_current_position() noexcept {
+  if (error()) { return error(); }
+  return first.get_current_position();
+}
+
+inline error_code simdjson_result<fallback::ondemand::object>::revert_position(fallback::ondemand::object_position position) noexcept {
+  if (error()) { return error(); }
+  return first.revert_position(position);
+}
+
 inline simdjson_result<bool> simdjson_result<fallback::ondemand::object>::is_empty() noexcept {
   if (error()) { return error(); }
   return first.is_empty();
@@ -85891,6 +108117,61 @@ simdjson_inline object_iterator::object_iterator(const value_iterator &_iter) no
   : iter{_iter}
 {}

+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline object_iterator::object_iterator(const value_iterator &_iter, object* _parent) noexcept
+  : parent{_parent}, iter{_iter}
+{
+  if (parent) parent->set_locked(true);
+}
+#endif
+
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline object_iterator::~object_iterator() noexcept
+{
+  if (parent) parent->set_locked(false);
+}
+
+simdjson_inline object_iterator::object_iterator(object_iterator&& other) noexcept
+  : has_been_referenced{other.has_been_referenced},
+    parent{other.parent},
+    iter{std::move(other.iter)}
+{
+  other.parent = nullptr;
+}
+
+simdjson_inline object_iterator& object_iterator::operator=(object_iterator&& other) noexcept {
+  if (this != &other)
+  {
+    if (parent)
+      parent->set_locked(false);
+    has_been_referenced = other.has_been_referenced;
+    parent = other.parent;
+    iter = std::move(other.iter);
+
+    other.parent = nullptr;
+  }
+  return *this;
+}
+
+simdjson_inline object_iterator::object_iterator(const object_iterator& other) noexcept
+  : has_been_referenced{other.has_been_referenced},
+    parent{nullptr},
+    iter{other.iter}
+{}
+
+simdjson_inline object_iterator& object_iterator::operator=(const object_iterator& other) noexcept {
+  if (this != &other)
+  {
+    if (parent)
+      parent->set_locked(false);
+    has_been_referenced = other.has_been_referenced;
+    parent = nullptr;
+    iter = other.iter;
+  }
+  return *this;
+}
+#endif
+
 simdjson_inline simdjson_result<field> object_iterator::operator*() noexcept {
 #if SIMDJSON_DEVELOPMENT_CHECKS
   // We must call * once per iteration.
@@ -86018,6 +108299,147 @@ simdjson_inline simdjson_result<fallback::ondemand::object_iterator> &simdjson_r

 #endif // SIMDJSON_GENERIC_ONDEMAND_OBJECT_ITERATOR_INL_H
 /* end file simdjson/generic/ondemand/object_iterator-inl.h for fallback */
+/* including simdjson/generic/ondemand/ranges-inl.h for fallback: #include "simdjson/generic/ondemand/ranges-inl.h" */
+/* begin file simdjson/generic/ondemand/ranges-inl.h for fallback */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/ranges.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array_iterator-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object_iterator-inl.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#if SIMDJSON_SUPPORTS_RANGES
+
+namespace simdjson {
+namespace fallback {
+namespace ondemand {
+
+//
+// array_range_iterator
+//
+
+simdjson_inline array_range_iterator::array_range_iterator(array_iterator iter) noexcept
+  : iter_{iter} {}
+
+simdjson_inline simdjson_result<value> array_range_iterator::operator*() const noexcept {
+  return *iter_;
+}
+
+simdjson_inline array_range_iterator& array_range_iterator::operator++() noexcept {
+  ++iter_;
+  return *this;
+}
+
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+simdjson_inline void array_range_iterator::operator++(int) noexcept {
+  ++*this;
+}
+SIMDJSON_POP_DISABLE_WARNINGS
+
+//
+// array_range
+//
+
+simdjson_inline array_range::array_range(array& arr) noexcept {
+  auto b = arr.begin();
+  if (b.error()) { error_ = b.error(); return; }
+  begin_ = b.value_unsafe();
+  end_ = arr.end().value_unsafe();
+}
+
+simdjson_inline array_range_iterator array_range::begin() noexcept {
+  return array_range_iterator(begin_);
+}
+
+simdjson_inline array_range_iterator array_range::end() noexcept {
+  return array_range_iterator(end_);
+}
+
+//
+// object_range_iterator
+//
+
+simdjson_inline object_range_iterator::object_range_iterator(object_iterator iter) noexcept
+  : iter_{iter} {}
+
+simdjson_inline simdjson_result<field> object_range_iterator::operator*() const noexcept {
+  return *iter_;
+}
+
+simdjson_inline object_range_iterator& object_range_iterator::operator++() noexcept {
+  ++iter_;
+  return *this;
+}
+
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+simdjson_inline void object_range_iterator::operator++(int) noexcept {
+  ++*this;
+}
+SIMDJSON_POP_DISABLE_WARNINGS
+
+
+//
+// object_range
+//
+
+simdjson_inline object_range::object_range(object& obj) noexcept {
+  auto b = obj.begin();
+  if (b.error()) { error_ = b.error(); return; }
+  begin_ = b.value_unsafe();
+  end_ = obj.end().value_unsafe();
+}
+
+simdjson_inline object_range_iterator object_range::begin() noexcept {
+  return object_range_iterator(begin_);
+}
+
+simdjson_inline object_range_iterator object_range::end() noexcept {
+  return object_range_iterator(end_);
+}
+
+//
+// Free functions
+//
+
+simdjson_inline array_range get_range(array& arr) noexcept {
+  return array_range(arr);
+}
+
+simdjson_inline object_range get_key_value_range(object& obj) noexcept {
+  return object_range(obj);
+}
+
+#if SIMDJSON_EXCEPTIONS
+simdjson_inline array_range get_range(simdjson_result<array> result) {
+  return array_range(result.value());
+}
+
+simdjson_inline object_range get_key_value_range(simdjson_result<object> result) {
+  return object_range(result.value());
+}
+#endif // SIMDJSON_EXCEPTIONS
+
+} // namespace ondemand
+} // namespace fallback
+} // namespace simdjson
+
+// Verify the range wrapper types satisfy the expected C++20 concepts.
+static_assert(std::input_iterator<simdjson::fallback::ondemand::array_range_iterator>);
+static_assert(std::input_iterator<simdjson::fallback::ondemand::object_range_iterator>);
+static_assert(std::ranges::input_range<simdjson::fallback::ondemand::array_range>);
+static_assert(std::ranges::input_range<simdjson::fallback::ondemand::object_range>);
+static_assert(std::ranges::view<simdjson::fallback::ondemand::array_range>);
+static_assert(std::ranges::view<simdjson::fallback::ondemand::object_range>);
+
+#endif // SIMDJSON_SUPPORTS_RANGES
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H
+/* end file simdjson/generic/ondemand/ranges-inl.h for fallback */
 /* including simdjson/generic/ondemand/parser-inl.h for fallback: #include "simdjson/generic/ondemand/parser-inl.h" */
 /* begin file simdjson/generic/ondemand/parser-inl.h for fallback */
 #ifndef SIMDJSON_GENERIC_ONDEMAND_PARSER_INL_H
@@ -86049,7 +108471,10 @@ simdjson_warn_unused simdjson_inline error_code parser::allocate(size_t new_capa

   // string_capacity copied from document::allocate
   _capacity = 0;
-  size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * new_capacity / 3 + SIMDJSON_PADDING, 64);
+  if(5 * (new_capacity / 3) + SIMDJSON_PADDING < SIMDJSON_PADDING) {
+    return CAPACITY; // overflow, only happen on legacy 32-bit systems with very large capacity
+  }
+  size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * (new_capacity / 3) + SIMDJSON_PADDING, 64);
   string_buf.reset(new (std::nothrow) uint8_t[string_capacity]);
 #if SIMDJSON_DEVELOPMENT_CHECKS
   start_positions.reset(new (std::nothrow) token_position[new_max_depth]);
@@ -86074,6 +108499,7 @@ simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate(p
   if (!json.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }

   json.remove_utf8_bom();
+  _document_len = json.length();

   // Allocate if needed
   if (capacity() < json.length() || !string_buf) {
@@ -86090,6 +108516,7 @@ simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate_a
   if (!json.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }

   json.remove_utf8_bom();
+  _document_len = json.length();

   // Allocate if needed
   if (capacity() < json.length() || !string_buf) {
@@ -86155,6 +108582,34 @@ simdjson_warn_unused simdjson_inline simdjson_result<json_iterator> parser::iter
   return json_iterator(reinterpret_cast<const uint8_t *>(json.data()), this);
 }

+inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size) noexcept {
+  if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
+  if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+    buf += 3;
+    len -= 3;
+  }
+  return document_stream(*this, buf, len, batch_size, false, stream_format::whitespace_delimited);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size) noexcept {
+  return iterate_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(padded_string_view s, size_t batch_size) noexcept {
+  if (!s.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
+  return iterate_many(s.data(), s.length(), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const padded_string &s, size_t batch_size) noexcept {
+  return iterate_many(padded_string_view(s), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const std::string &s, size_t batch_size) noexcept {
+  return iterate_many(padded_string_view(s), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size) noexcept {
+  return iterate_many(pad(s), batch_size);
+}
+
+#ifndef SIMDJSON_DISABLE_DEPRECATED_API
+SIMDJSON_PUSH_DISABLE_WARNINGS
+SIMDJSON_DISABLE_DEPRECATED_WARNING
 inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept {
   // Warning: no check is done on the buffer padding. We trust the user.
   if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
@@ -86162,8 +108617,11 @@ inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf,
     buf += 3;
     len -= 3;
   }
-  if(allow_comma_separated && batch_size < len) { batch_size = len; }
-  return document_stream(*this, buf, len, batch_size, allow_comma_separated);
+  // Map allow_comma_separated to stream_format::comma_delimited
+  if (allow_comma_separated) {
+    return document_stream(*this, buf, len, batch_size, false, stream_format::comma_delimited);
+  }
+  return document_stream(*this, buf, len, batch_size, false, stream_format::whitespace_delimited);
 }

 inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept {
@@ -86183,6 +108641,51 @@ inline simdjson_result<document_stream> parser::iterate_many(const std::string &
 inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size, bool allow_comma_separated) noexcept {
   return iterate_many(pad(s), batch_size, allow_comma_separated);
 }
+SIMDJSON_POP_DISABLE_WARNINGS
+#endif // SIMDJSON_DISABLE_DEPRECATED_API
+
+inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+  if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
+  if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+    buf += 3;
+    len -= 3;
+  }
+  if (format == stream_format::comma_delimited_array) {
+    // Strip leading JSON whitespace.
+    while (len > 0 && (buf[0] == ' ' || buf[0] == '\t' || buf[0] == '\n' || buf[0] == '\r')) {
+      buf++; len--;
+    }
+    // Expect the opening '['.
+    if (len == 0 || buf[0] != '[') { return TAPE_ERROR; }
+    buf++; len--;
+    // Strip trailing JSON whitespace.
+    while (len > 0 && (buf[len-1] == ' ' || buf[len-1] == '\t' || buf[len-1] == '\n' || buf[len-1] == '\r')) {
+      len--;
+    }
+    // Expect the closing ']'.
+    if (len == 0 || buf[len-1] != ']') { return TAPE_ERROR; }
+    len--;
+    // Fall through to comma_delimited over the array contents.
+    format = stream_format::comma_delimited;
+  }
+  return document_stream(*this, buf, len, batch_size, false, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+  return iterate_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(padded_string_view s, size_t batch_size, stream_format format) noexcept {
+  if (!s.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
+  return iterate_many(s.data(), s.length(), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const std::string &s, size_t batch_size, stream_format format) noexcept {
+  return iterate_many(padded_string_view(s), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size, stream_format format) noexcept {
+  return iterate_many(pad(s), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const padded_string &s, size_t batch_size, stream_format format) noexcept {
+  return iterate_many(padded_string_view(s), batch_size, format);
+}
 simdjson_pure simdjson_inline size_t parser::capacity() const noexcept {
   return _capacity;
 }
@@ -86590,6 +109093,27 @@ namespace simdjson {
 namespace fallback {
 namespace ondemand {

+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+simdjson_inline bool raw_json_string_is_quote_terminated(const uint8_t *json, uint32_t max_len) noexcept {
+  bool escaping{false};
+  for (uint32_t i = 1; i < max_len; i++) {
+    switch (json[i]) {
+      case '"':
+        if (!escaping) { return true; }
+        escaping = false;
+        break;
+      case '\\':
+        escaping = !escaping;
+        break;
+      default:
+        escaping = false;
+        break;
+    }
+  }
+  return false;
+}
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+
 simdjson_inline value_iterator::value_iterator(
   json_iterator *json_iter,
   depth_t depth,
@@ -86977,6 +109501,22 @@ simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iter
   return raw_json_string(key);
 }

+simdjson_warn_unused simdjson_inline error_code value_iterator::field_key_with_length(raw_json_string &key, std::size_t &len) noexcept {
+  assert_at_next();
+
+  const uint8_t *k = _json_iter->return_current_and_advance();
+  if (*(k++) != '"') { return report_error(TAPE_ERROR, "Object key is not a string"); }
+  // After return_current_and_advance(), the current token is the ':' that follows
+  // the key. The closing quote sits just before it (only JSON whitespace may
+  // intervene), so step back from the ':' to the closing quote to get the length.
+  // In minified JSON this is a single back-step.
+  const char *q = reinterpret_cast<const char *>(_json_iter->peek());
+  do { --q; } while (*q != '"');
+  key = raw_json_string(k);
+  len = static_cast<std::size_t>(q - reinterpret_cast<const char *>(k));
+  return SUCCESS;
+}
+
 simdjson_warn_unused simdjson_inline error_code value_iterator::field_value() noexcept {
   assert_at_next();

@@ -87094,7 +109634,7 @@ simdjson_warn_unused simdjson_inline error_code value_iterator::get_string(strin
   auto saved_string_buf_loc = _json_iter->string_buf_loc();
   auto err = get_string(allow_replacement).get(content);
   if (err) { return err; }
-  receiver = content;
+  internal::assign_utf8(receiver, content);
   // Restore the string buffer location, effectively discarding any temporary string storage
   _json_iter->string_buf_loc() = saved_string_buf_loc;
   return SUCCESS;
@@ -87105,6 +109645,15 @@ simdjson_warn_unused simdjson_inline simdjson_result<std::string_view> value_ite
 simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iterator::get_raw_json_string() noexcept {
   auto json = peek_scalar("string");
   if (*json != '"') { return incorrect_type_error("Not a string"); }
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  if (_json_iter->allow_incomplete_json()) {
+    const size_t remaining_input_length = _json_iter->remaining_input_length(json);
+    const uint32_t max_len = remaining_input_length < peek_start_length() ? uint32_t(remaining_input_length) : peek_start_length();
+    if (!raw_json_string_is_quote_terminated(json, max_len)) {
+      return STRING_ERROR;
+    }
+  }
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
   advance_scalar("string");
   return raw_json_string(json+1);
 }
@@ -87138,6 +109687,16 @@ simdjson_warn_unused simdjson_inline simdjson_result<double> value_iterator::get
   if(result.error() == SUCCESS) { advance_non_root_scalar("double"); }
   return result;
 }
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_float() noexcept {
+  auto result = numberparsing::parse_float(peek_non_root_scalar("float"));
+  if(result.error() == SUCCESS) { advance_non_root_scalar("float"); }
+  return result;
+}
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_float_in_string() noexcept {
+  auto result = numberparsing::parse_float_in_string(peek_non_root_scalar("float"));
+  if(result.error() == SUCCESS) { advance_non_root_scalar("float"); }
+  return result;
+}
 simdjson_warn_unused simdjson_inline simdjson_result<bool> value_iterator::get_bool() noexcept {
   auto result = parse_bool(peek_non_root_scalar("bool"));
   if(result.error() == SUCCESS) { advance_non_root_scalar("bool"); }
@@ -87240,7 +109799,7 @@ simdjson_warn_unused simdjson_inline error_code value_iterator::get_root_string(
   auto saved_string_buf_loc = _json_iter->string_buf_loc();
   auto err = get_root_string(check_trailing, allow_replacement).get(content);
   if (err) { return err; }
-  receiver = content;
+  internal::assign_utf8(receiver, content);
   // Restore the string buffer location, effectively discarding any temporary string storage
   _json_iter->string_buf_loc() = saved_string_buf_loc;
   return SUCCESS;
@@ -87252,6 +109811,15 @@ simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iter
   auto json = peek_scalar("string");
   if (*json != '"') { return incorrect_type_error("Not a string"); }
   if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  if (_json_iter->allow_incomplete_json()) {
+    const size_t remaining_input_length = _json_iter->remaining_input_length(json);
+    const uint32_t max_len = remaining_input_length < peek_root_length() ? uint32_t(remaining_input_length) : peek_root_length();
+    if (!raw_json_string_is_quote_terminated(json, max_len)) {
+      return STRING_ERROR;
+    }
+  }
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
   advance_scalar("string");
   return raw_json_string(json+1);
 }
@@ -87361,6 +109929,43 @@ simdjson_warn_unused simdjson_inline simdjson_result<double> value_iterator::get
   return result;
 }

+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_root_float(bool check_trailing) noexcept {
+  auto max_len = peek_root_length();
+  auto json = peek_root_scalar("float");
+  // We use the same buffer size as get_root_double: the number of significant
+  // digits that matter is smaller for binary32, but the JSON document may still
+  // spell out a long number that we must parse (and round) faithfully.
+  uint8_t tmpbuf[1074+8+1+1]; // +1 for null termination.
+  tmpbuf[1074+8+1] = '\0'; // make sure that buffer is always null terminated.
+  if (!_json_iter->copy_to_buffer(json, max_len, tmpbuf, 1074+8+1)) {
+    logger::log_error(*_json_iter, start_position(), depth(), "Root number more than 1082 characters");
+    return NUMBER_ERROR;
+  }
+  auto result = numberparsing::parse_float(tmpbuf);
+  if(result.error() == SUCCESS) {
+    if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+    advance_root_scalar("float");
+  }
+  return result;
+}
+
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_root_float_in_string(bool check_trailing) noexcept {
+  auto max_len = peek_root_length();
+  auto json = peek_root_scalar("float");
+  uint8_t tmpbuf[1074+8+1+1]; // +1 for null termination.
+  tmpbuf[1074+8+1] = '\0'; // make sure that buffer is always null terminated.
+  if (!_json_iter->copy_to_buffer(json, max_len, tmpbuf, 1074+8+1)) {
+    logger::log_error(*_json_iter, start_position(), depth(), "Root number more than 1082 characters");
+    return NUMBER_ERROR;
+  }
+  auto result = numberparsing::parse_float_in_string(tmpbuf);
+  if(result.error() == SUCCESS) {
+    if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+    advance_root_scalar("float");
+  }
+  return result;
+}
+
 simdjson_warn_unused simdjson_inline simdjson_result<bool> value_iterator::get_root_bool(bool check_trailing) noexcept {
   auto max_len = peek_root_length();
   auto json = peek_root_scalar("bool");
@@ -87589,6 +110194,21 @@ simdjson_inline void value_iterator::move_at_container_start() noexcept {
   _json_iter->token.set_position(_start_position + 1);
 }

+simdjson_inline void value_iterator::reenter_at(token_position position, depth_t depth) noexcept {
+  // Unlike reenter_child(), this does not require the live depth to be
+  // exactly one level shallower than depth, nor does it validate against
+  // the parser's per-depth container-start bookkeeping: neither holds in
+  // general for a caller-supplied snapshot (see object_position). What
+  // must still always hold, regardless of what was captured or how far
+  // the live iterator has since moved, is that position and depth are
+  // themselves sane values -- this is the same bound reenter_child()
+  // itself applies unconditionally.
+  SIMDJSON_ASSUME(position != nullptr);
+  SIMDJSON_ASSUME(depth >= 1 && depth < INT32_MAX);
+  _json_iter->_depth = depth;
+  _json_iter->token.set_position(position);
+}
+
 simdjson_inline simdjson_result<bool> value_iterator::reset_array() noexcept {
   if(error()) { return error(); }
   move_at_container_start();
@@ -89041,16 +111661,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
 }
 #endif

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
-                                uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
-  return _addcarry_u64(0, value1, value2,
-                       reinterpret_cast<unsigned __int64 *>(result));
-#else
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-#endif
-}

 } // unnamed namespace
 } // namespace haswell
@@ -89610,7 +112220,7 @@ simdjson_inline escaping escaping::copy_and_find(const uint8_t *src, uint8_t *ds
 /* end file simdjson/haswell/begin.h */
 /* including simdjson/generic/ondemand/amalgamated.h for haswell: #include "simdjson/generic/ondemand/amalgamated.h" */
 /* begin file simdjson/generic/ondemand/amalgamated.h for haswell */
-#if defined(SIMDJSON_CONDITIONAL_INCLUDE) && !defined(SIMDJSON_GENERIC_BUILDER_DEPENDENCIES_H)
+#if defined(SIMDJSON_CONDITIONAL_INCLUDE) && !defined(SIMDJSON_GENERIC_ONDEMAND_DEPENDENCIES_H)
 #error simdjson/generic/ondemand/dependencies.h must be included before simdjson/generic/ondemand/amalgamated.h!
 #endif

@@ -89659,6 +112269,13 @@ class token_iterator;
 class value;
 class value_iterator;

+#if SIMDJSON_SUPPORTS_RANGES
+class array_range;
+class array_range_iterator;
+class object_range;
+class object_range_iterator;
+#endif // SIMDJSON_SUPPORTS_RANGES
+
 } // namespace ondemand
 } // namespace haswell
 } // namespace simdjson
@@ -89691,6 +112308,9 @@ template <> struct is_builtin_deserializable<haswell::ondemand::object> : std::t
 template <> struct is_builtin_deserializable<haswell::ondemand::value> : std::true_type {};
 template <> struct is_builtin_deserializable<haswell::ondemand::raw_json_string> : std::true_type {};
 template <> struct is_builtin_deserializable<std::string_view> : std::true_type {};
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template <> struct is_builtin_deserializable<std::u8string_view> : std::true_type {};
+#endif // SIMDJSON_SUPPORTS_CHAR8_T

 template <typename T>
 concept is_builtin_deserializable_v = is_builtin_deserializable<T>::value;
@@ -89708,6 +112328,10 @@ concept nothrow_custom_deserializable = nothrow_tag_invocable<deserialize_tag, V
 template <typename T, typename ValT = haswell::ondemand::value>
 concept nothrow_deserializable = nothrow_custom_deserializable<T, ValT> || is_builtin_deserializable_v<T>;

+// True iff get<T>() on ValT cannot throw: no custom tag_invoke, or that tag_invoke is noexcept.
+template <typename T, typename ValT = haswell::ondemand::value>
+concept nothrow_gettable = !custom_deserializable<T, ValT> || nothrow_custom_deserializable<T, ValT>;
+
 /// Deserialize Tag
 inline constexpr struct deserialize_tag {
   using array_type = haswell::ondemand::array;
@@ -89922,6 +112546,17 @@ public:
    */
   simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> field_key() noexcept;

+  /**
+   * Get the current field's key together with its raw byte length.
+   *
+   * Like field_key(), but also returns the number of raw key bytes (the distance
+   * from the first key byte to the closing quote). The length is recovered from
+   * the structural index -- the next structural token is the ':' -- by stepping
+   * back over any whitespace to the closing quote, avoiding a forward SIMD scan
+   * for the closing quote. Leaves the iterator positioned exactly as field_key().
+   */
+  simdjson_warn_unused simdjson_inline error_code field_key_with_length(raw_json_string &key, std::size_t &len) noexcept;
+
   /**
    * Pass the : in the field and move to its value.
    */
@@ -90074,6 +112709,8 @@ public:
   simdjson_warn_unused simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_warn_unused simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_warn_unused simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<bool> get_bool() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<bool> is_null() noexcept;
   simdjson_warn_unused simdjson_inline bool is_negative() noexcept;
@@ -90092,6 +112729,8 @@ public:
   simdjson_warn_unused simdjson_inline simdjson_result<int64_t> get_root_int64_in_string(bool check_trailing) noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<double> get_root_double(bool check_trailing) noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<double> get_root_double_in_string(bool check_trailing) noexcept;
+  simdjson_warn_unused simdjson_inline simdjson_result<float> get_root_float(bool check_trailing) noexcept;
+  simdjson_warn_unused simdjson_inline simdjson_result<float> get_root_float_in_string(bool check_trailing) noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<bool> get_root_bool(bool check_trailing) noexcept;
   simdjson_warn_unused simdjson_inline bool is_root_negative() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<bool> is_root_integer(bool check_trailing) noexcept;
@@ -90227,6 +112866,15 @@ protected:

   /** @copydoc error_code json_iterator::position() const noexcept; */
   simdjson_inline token_position position() const noexcept;
+  /**
+   * Move the live iterator directly to the given position and depth, without
+   * validating against the parser's per-depth container-start bookkeeping
+   * (unlike json_iterator::reenter_child()). Used to restore a previously
+   * captured mid-container position (see object::revert_position()): that
+   * bookkeeping only tracks each container's own start, not every position
+   * a caller might later capture and revert to, so it does not apply here.
+   */
+  simdjson_inline void reenter_at(token_position position, depth_t depth) noexcept;
   /** @copydoc error_code json_iterator::end_position() const noexcept; */
   simdjson_inline token_position last_position() const noexcept;
   /** @copydoc error_code json_iterator::end_position() const noexcept; */
@@ -90295,9 +112943,12 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+   * Supported types: object, array, raw_json_string, string_view, u8string_view (C++20), uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool
    *
-   * You may use get_double(), get_bool(), get_uint64(), get_int64(),
+   * You may use get_double(), get_float(), get_bool(), get_uint64(), get_int64(),
+   * get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(),
+   * get_float32(), get_float64(),
    * get_object(), get_array(), get_raw_json_string(), or get_string() instead.
    * When SIMDJSON_SUPPORTS_CONCEPTS is set, custom types are also supported.
    *
@@ -90307,7 +112958,7 @@ public:
   template <typename T>
   simdjson_inline simdjson_result<T> get()
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+    noexcept(nothrow_gettable<T, value>)
 #else
     noexcept
 #endif
@@ -90322,7 +112973,8 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+   * Supported types: object, array, raw_json_string, string_view, u8string_view (C++20), uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool
    * If the macro SIMDJSON_SUPPORTS_CONCEPTS is set, then custom types are also supported.
    *
    * @param out This is set to a value of the given type, parsed from the JSON. If there is an error, this may not be initialized.
@@ -90332,7 +112984,7 @@ public:
   template <typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out)
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+    noexcept(nothrow_gettable<T, value>)
 #else
     noexcept
 #endif
@@ -90360,7 +113012,7 @@ public:
       "And you do not seem to have added support for it. Indeed, we have that "
       "simdjson::custom_deserializable<T> is false and the type T is not a default type "
       "such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-      "int64_t, double, or bool.");
+      "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
     static_cast<void>(out); // to get rid of unused errors
     return UNINITIALIZED;
   }
@@ -90369,7 +113021,8 @@ public:
     // immediately fail.
     static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
       "The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-      "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+      "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+      "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
       " get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
       " You may also add support for custom types, see our documentation.");
     static_cast<void>(out); // to get rid of unused errors
@@ -90447,6 +113100,50 @@ public:
    */
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;

+  /**
+   * Cast this JSON value to a 16-bit unsigned integer.
+   *
+   * Calls get_uint64() and checks that the result fits in a uint16_t.
+   *
+   * @returns A 16-bit unsigned integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint16_t.
+   */
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+
+  /**
+   * Cast this JSON value to a 16-bit signed integer.
+   *
+   * Calls get_int64() and checks that the result fits in an int16_t.
+   *
+   * @returns A 16-bit signed integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int16_t.
+   */
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+
+  /**
+   * Cast this JSON value to an 8-bit unsigned integer.
+   *
+   * Calls get_uint64() and checks that the result fits in a uint8_t.
+   *
+   * @returns An 8-bit unsigned integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint8_t.
+   */
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+
+  /**
+   * Cast this JSON value to an 8-bit signed integer.
+   *
+   * Calls get_int64() and checks that the result fits in an int8_t.
+   *
+   * @returns An 8-bit signed integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int8_t.
+   */
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
+
   /**
    * Cast this JSON value to a double.
    *
@@ -90463,6 +113160,53 @@ public:
    */
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;

+  /**
+   * Cast this JSON value to a float (binary32).
+   *
+   * The value is rounded directly to binary32: it is the float nearest to the
+   * JSON number. Note that this may differ from get_double() followed by a cast
+   * to float, which rounds twice.
+   *
+   * @returns A float.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+
+  /**
+   * Cast this JSON value (inside string) to a float (binary32).
+   *
+   * @returns A float.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  /**
+   * Cast this JSON value to a std::float32_t (C++23).
+   *
+   * Same as get_float(): the value is rounded directly to binary32.
+   *
+   * @returns A std::float32_t.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  /**
+   * Cast this JSON value to a std::float64_t (C++23).
+   *
+   * Same as get_double().
+   *
+   * @returns A std::float64_t.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   */
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
+
   /**
    * Cast this JSON value to a string.
    *
@@ -90490,6 +113234,26 @@ public:
    */
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;

+#if SIMDJSON_SUPPORTS_CHAR8_T
+  /**
+   * Cast this JSON value to a C++20 UTF-8 string.
+   *
+   * The string is guaranteed to be valid UTF-8.
+   *
+   * Equivalent to get<std::u8string_view>(). This is a zero-copy alias of
+   * get_string(): the very same bytes are returned, viewed as char8_t.
+   *
+   * Important: a value should be consumed once. Calling get_u8string() twice on the same
+   * value is an error.
+   *
+   * @param allow_replacement Whether to allow a replacement character for unmatched surrogate pairs.
+   * @returns An UTF-8 string. The string is stored in the parser and will be invalidated the next
+   *          time it parses a document or when it is destroyed.
+   * @returns INCORRECT_TYPE if the JSON value is not a string.
+   */
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
   /**
    * Attempts to fill the provided std::string reference with the parsed value of the current string.
    *
@@ -90577,7 +113341,7 @@ public:
   /**
    * Cast this JSON value to an unsigned integer.
    *
-   * @returns A signed 64-bit integer.
+   * @returns A unsigned 64-bit integer.
    * @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit unsigned integer.
    */
   simdjson_inline operator uint64_t() noexcept(false);
@@ -91042,9 +113806,24 @@ public:
   simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template <typename string_type>
   simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -91052,9 +113831,19 @@ public:
   simdjson_inline simdjson_result<bool> get_bool() noexcept;
   simdjson_inline simdjson_result<bool> is_null() noexcept;

-  template<typename T> simdjson_inline simdjson_result<T> get() noexcept;
+  template<typename T> simdjson_inline simdjson_result<T> get()
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, haswell::ondemand::value>);
+#else
+    noexcept;
+#endif

-  template<typename T> simdjson_inline error_code get(T &out) noexcept;
+  template<typename T> simdjson_inline error_code get(T &out)
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, haswell::ondemand::value>);
+#else
+    noexcept;
+#endif

 #if SIMDJSON_EXCEPTIONS
   template <class T>
@@ -91385,6 +114174,7 @@ protected:
   token_position _position{};

   friend class json_iterator;
+  friend class document_stream;
   friend class value_iterator;
   friend class object;
   template <typename... Args>
@@ -91476,6 +114266,9 @@ protected:
    * value of this attribute.
    */
   bool _streaming{false};
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  bool _allow_incomplete_json{false};
+#endif

 public:
   simdjson_inline json_iterator() noexcept = default;
@@ -91500,6 +114293,10 @@ public:
    * start_root_array() and start_root_object().
    */
   simdjson_inline bool streaming() const noexcept;
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  simdjson_inline bool allow_incomplete_json() const noexcept;
+  simdjson_inline size_t remaining_input_length(const uint8_t *json) const noexcept;
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON

   /**
    * Get the root value iterator
@@ -92379,33 +115176,87 @@ public:
    * @param batch_size The batch size to use. MUST be larger than the largest document. The sweet
    *                   spot is cache-related: small enough to fit in cache, yet big enough to
    *                   parse as many documents as possible in one tight loop.
-   *                   Defaults to 10MB, which has been a reasonable sweet spot in our tests.
-   * @param allow_comma_separated (defaults on false) This allows a mode where the documents are
-   *                   separated by commas instead of whitespace. It comes with a performance
-   *                   penalty because the entire document is indexed at once (and the document must be
-   *                   less than 4 GB), and there is no multithreading. In this mode, the batch_size parameter
-   *                   is effectively ignored, as it is set to at least the document size.
+   *                   Defaults to 1MB, which has been a reasonable sweet spot in our tests.
+   * @param allow_comma_separated @deprecated Use stream_format::comma_delimited instead.
+   *                   When true, maps internally to stream_format::comma_delimited.
+   *                   Defaults to false.
    * @return The stream, or an error. An empty input will yield 0 documents rather than an EMPTY error. Errors:
    *         - MEMALLOC if the parser does not have enough capacity and memory allocation fails
    *         - CAPACITY if the parser does not have enough capacity and batch_size > max_capacity.
    *         - other json errors if parsing fails. You should not rely on these errors to always the same for the
    *           same document: they may vary under runtime dispatch (so they may vary depending on your system and hardware).
    */
-  inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
-  inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
-  inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
-  inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size)
+  inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size)
     the string might be automatically padded with up to SIMDJSON_PADDING whitespace characters */
-  inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
-  inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
+  inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @private An rvalue input is destroyed at the end of the full-expression, while the
+   * returned document_stream only holds a pointer to it: iterating the stream would then
+   * read freed memory. These deleted overloads also catch a std::string_view argument,
+   * which would otherwise convert implicitly to a padded_string temporary. */
+  inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size = DEFAULT_BATCH_SIZE) = delete;// unsafe
+  /** @private @overload iterate_many(const std::string &&s, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size = DEFAULT_BATCH_SIZE) = delete;// unsafe
   /** @private We do not want to allow implicit conversion from C string to std::string. */
   simdjson_result<document_stream> iterate_many(const char *buf, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept = delete;

+#ifndef SIMDJSON_DISABLE_DEPRECATED_API
+  /**
+   * @deprecated Use iterate_many with stream_format::comma_delimited instead.
+   */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @private Deleted for the same reason as iterate_many(const std::string &&s, size_t batch_size). */
+  inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size, bool allow_comma_separated) = delete;// unsafe
+  /** @private @overload iterate_many(const std::string &&s, size_t batch_size, bool allow_comma_separated) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size, bool allow_comma_separated) = delete;// unsafe
+#endif // SIMDJSON_DISABLE_DEPRECATED_API
+
+  /**
+   * Parse a stream of JSON documents with explicit format specification.
+   *
+   * @param buf The concatenated JSON documents.
+   * @param len The length of the buffer.
+   * @param batch_size The batch size to use.
+   * @param format The stream format (whitespace_delimited, json_sequence, or comma_delimited).
+   * @return A stream of documents, or an error.
+   */
+  inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(padded_string_view s, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format)
+   *
+   * The string is padded in place if needed (see simdjson::pad), as with iterate_many(std::string &s, size_t batch_size).
+   */
+  inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size, stream_format format) noexcept;
+  /** @private Deleted for the same reason as iterate_many(const std::string &&s, size_t batch_size). */
+  inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+  /** @private @overload iterate_many(const std::string &&s, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+
   /** The capacity of this parser (the largest document it can process). */
   simdjson_pure simdjson_inline size_t capacity() const noexcept;
   /** The maximum capacity of this parser (the largest document it is allowed to process). */
@@ -92533,6 +115384,7 @@ private:
   size_t _capacity{0};
   size_t _max_capacity;
   size_t _max_depth{DEFAULT_MAX_DEPTH};
+  size_t _document_len{0};
   std::unique_ptr<uint8_t[]> string_buf{};

 #if SIMDJSON_DEVELOPMENT_CHECKS
@@ -92595,8 +115447,19 @@ public:
    * Begin array iteration.
    *
    * Part of the std::iterable interface.
+   *
+   * With SIMDJSON_DEVELOPMENT_CHECKS, the iterator locks this array while it is
+   * alive, so that reentrant access (at(), count_elements(), reset(), ...) is
+   * reported as OUT_OF_ORDER_ITERATION.
+   */
+  simdjson_inline simdjson_result<array_iterator> begin() & noexcept;
+  /**
+   * Begin iteration over a temporary array, e.g., `v.get_array().begin()`.
+   *
+   * The iterator does not depend on the array instance and may outlive it, so
+   * it does not lock it.
    */
-  simdjson_inline simdjson_result<array_iterator> begin() noexcept;
+  simdjson_inline simdjson_result<array_iterator> begin() && noexcept;
   /**
    * Sentinel representing the end of the array.
    *
@@ -92727,7 +115590,7 @@ public:
    */
   template <typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out)
-     noexcept(custom_deserializable<T, array> ? nothrow_custom_deserializable<T, array> : true) {
+     noexcept(nothrow_gettable<T, array>) {
     static_assert(custom_deserializable<T, array>);
     return deserialize(*this, out);
   }
@@ -92739,7 +115602,7 @@ public:
    */
   template <typename T>
   simdjson_inline simdjson_result<T> get()
-    noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+    noexcept(nothrow_gettable<T, array>)
   {
     static_assert(std::is_default_constructible<T>::value, "The specified type is not default constructible.");
     T out{};
@@ -92795,6 +115658,10 @@ protected:
    * iter.is_alive() == false indicates iteration is complete.
    */
   value_iterator iter{};
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  bool locked{false};
+  simdjson_inline void set_locked(bool _locked) noexcept;
+#endif

   friend class value;
   friend class document;
@@ -92816,7 +115683,8 @@ public:
   simdjson_inline simdjson_result(error_code error) noexcept; ///< @private
   simdjson_inline simdjson_result() noexcept = default;

-  simdjson_inline simdjson_result<haswell::ondemand::array_iterator> begin() noexcept;
+  simdjson_inline simdjson_result<haswell::ondemand::array_iterator> begin() & noexcept;
+  simdjson_inline simdjson_result<haswell::ondemand::array_iterator> begin() && noexcept;
   simdjson_inline simdjson_result<haswell::ondemand::array_iterator> end() noexcept;
   inline simdjson_result<size_t> count_elements() & noexcept;
   inline simdjson_result<bool> is_empty() & noexcept;
@@ -92836,7 +115704,7 @@ public:
   // TODO: move this code into object-inl.h

   template<typename T>
-  simdjson_inline simdjson_result<T> get() noexcept {
+  simdjson_inline simdjson_result<T> get() noexcept(nothrow_gettable<T, haswell::ondemand::array>) {
     if (error()) { return error(); }
     if constexpr (std::is_same_v<T, haswell::ondemand::array>) {
       return first;
@@ -92844,7 +115712,7 @@ public:
     return first.get<T>();
   }
   template<typename T>
-  simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept {
+  simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept(nothrow_gettable<T, haswell::ondemand::array>) {
     if (error()) { return error(); }
     if constexpr (std::is_same_v<T, haswell::ondemand::array>) {
       out = first;
@@ -92896,6 +115764,15 @@ public:
   /** Create a new, invalid array iterator. */
   simdjson_inline array_iterator() noexcept = default;

+#if SIMDJSON_DEVELOPMENT_CHECKS
+  simdjson_inline ~array_iterator() noexcept;
+
+  simdjson_inline array_iterator(array_iterator&&) noexcept;
+  simdjson_inline array_iterator& operator=(array_iterator&&) noexcept;
+  simdjson_inline array_iterator(const array_iterator&) noexcept;
+  simdjson_inline array_iterator& operator=(const array_iterator&) noexcept;
+#endif
+
   //
   // Iterator interface
   //
@@ -92938,6 +115815,9 @@ public:
 private:
 #if SIMDJSON_DEVELOPMENT_CHECKS
    bool has_been_referenced{false};
+   array* parent{nullptr};
+
+   simdjson_inline array_iterator(const value_iterator &_iter, array* _parent) noexcept;
 #endif
   value_iterator iter{};

@@ -93037,14 +115917,14 @@ public:
   /**
    * Cast this JSON value to an unsigned integer.
    *
-   * @returns A signed 64-bit integer.
+   * @returns A unsigned 64-bit integer.
    * @returns INCORRECT_TYPE If the JSON value is not a 64-bit unsigned integer.
    */
   simdjson_inline simdjson_result<uint64_t> get_uint64() noexcept;
   /**
    * Cast this JSON value (inside string) to an unsigned integer.
    *
-   * @returns A signed 64-bit integer.
+   * @returns A unsigned 64-bit integer.
    * @returns INCORRECT_TYPE If the JSON value is not a 64-bit unsigned integer.
    */
   simdjson_inline simdjson_result<uint64_t> get_uint64_in_string() noexcept;
@@ -93082,6 +115962,46 @@ public:
    * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int32_t.
    */
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  /**
+   * Cast this JSON value to a 16-bit unsigned integer.
+   *
+   * Calls get_uint64() and checks that the result fits in a uint16_t.
+   *
+   * @returns A 16-bit unsigned integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint16_t.
+   */
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  /**
+   * Cast this JSON value to a 16-bit signed integer.
+   *
+   * Calls get_int64() and checks that the result fits in an int16_t.
+   *
+   * @returns A 16-bit signed integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int16_t.
+   */
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  /**
+   * Cast this JSON value to an 8-bit unsigned integer.
+   *
+   * Calls get_uint64() and checks that the result fits in a uint8_t.
+   *
+   * @returns An 8-bit unsigned integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint8_t.
+   */
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  /**
+   * Cast this JSON value to an 8-bit signed integer.
+   *
+   * Calls get_int64() and checks that the result fits in an int8_t.
+   *
+   * @returns An 8-bit signed integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int8_t.
+   */
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   /**
    * Cast this JSON value to a double.
    *
@@ -93097,6 +116017,53 @@ public:
    * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
    */
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+
+  /**
+   * Cast this JSON value to a float (binary32).
+   *
+   * The value is rounded directly to binary32: it is the float nearest to the
+   * JSON number. Note that this may differ from get_double() followed by a cast
+   * to float, which rounds twice.
+   *
+   * @returns A float.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+
+  /**
+   * Cast this JSON value (inside string) to a float (binary32).
+   *
+   * @returns A float.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  /**
+   * Cast this JSON value to a std::float32_t (C++23).
+   *
+   * Same as get_float(): the value is rounded directly to binary32.
+   *
+   * @returns A std::float32_t.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  /**
+   * Cast this JSON value to a std::float64_t (C++23).
+   *
+   * Same as get_double().
+   *
+   * @returns A std::float64_t.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   */
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   /**
    * Cast this JSON value to a string.
    *
@@ -93110,6 +116077,24 @@ public:
    * @returns INCORRECT_TYPE if the JSON value is not a string.
    */
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  /**
+   * Cast this JSON value to a C++20 UTF-8 string.
+   *
+   * The string is guaranteed to be valid UTF-8.
+   *
+   * Equivalent to get<std::u8string_view>(). This is a zero-copy alias of
+   * get_string(): the very same bytes are returned, viewed as char8_t.
+   *
+   * Important: Calling get_u8string() twice on the same document is an error.
+   *
+   * @param allow_replacement Whether to allow a replacement character for unmatched surrogate pairs.
+   * @returns An UTF-8 string. The string is stored in the parser and will be invalidated the next
+   *          time it parses a document or when it is destroyed.
+   * @returns INCORRECT_TYPE if the JSON value is not a string.
+   */
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   /**
    * Attempts to fill the provided std::string reference with the parsed value of the current string.
    *
@@ -93180,9 +116165,12 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool
    *
-   * You may use get_double(), get_bool(), get_uint64(), get_int64(),
+   * You may use get_double(), get_float(), get_bool(), get_uint64(), get_int64(),
+   * get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(),
+   * get_float32(), get_float64(),
    * get_object(), get_array(), get_raw_json_string(), or get_string() instead.
    *
    * @returns A value of the given type, parsed from the JSON.
@@ -93191,7 +116179,7 @@ public:
   template <typename T>
   simdjson_inline simdjson_result<T> get() &
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    noexcept(nothrow_gettable<T, document>)
 #else
     noexcept
 #endif
@@ -93214,7 +116202,7 @@ public:
   template<typename T>
   simdjson_inline simdjson_result<T> get() &&
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    noexcept(nothrow_gettable<T, document>)
 #else
     noexcept
 #endif
@@ -93226,7 +116214,8 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool, value
+   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool, value
    *
    * Be mindful that the document instance must remain in scope while you are accessing object, array and value instances.
    *
@@ -93237,7 +116226,7 @@ public:
   template<typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out) &
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    noexcept(nothrow_gettable<T, document>)
 #else
     noexcept
 #endif
@@ -93250,7 +116239,7 @@ public:
         "And you do not seem to have added support for it. Indeed, we have that "
         "simdjson::custom_deserializable<T> is false and the type T is not a default type "
         "such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-        "int64_t, double, or bool.");
+        "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
       static_cast<void>(out); // to get rid of unused errors
       return UNINITIALIZED;
     }
@@ -93259,7 +116248,8 @@ public:
     // immediately fail.
     static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
       "The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-      "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+      "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+      "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
       " get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
       " You may also add support for custom types, see our documentation.");
     static_cast<void>(out); // to get rid of unused errors
@@ -93268,7 +116258,12 @@ public:
   }

   /** @overload template<typename T> error_code get(T &out) & noexcept */
-  template<typename T> simdjson_deprecated simdjson_inline error_code get(T &out) && noexcept;
+  template<typename T> simdjson_deprecated simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, document>);
+#else
+    noexcept;
+#endif

 #if SIMDJSON_EXCEPTIONS
   /**
@@ -93302,24 +116297,24 @@ public:
   /**
    * Cast this JSON value to an unsigned integer.
    *
-   * @returns A signed 64-bit integer.
+   * @returns A unsigned 64-bit integer.
    * @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit unsigned integer.
    */
-  simdjson_inline operator uint64_t() noexcept(false);
+  explicit simdjson_inline operator uint64_t() noexcept(false);
   /**
    * Cast this JSON value to a signed integer.
    *
    * @returns A signed 64-bit integer.
    * @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit integer.
    */
-  simdjson_inline operator int64_t() noexcept(false);
+  explicit simdjson_inline operator int64_t() noexcept(false);
   /**
    * Cast this JSON value to a double.
    *
    * @returns A double.
    * @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a valid floating-point number.
    */
-  simdjson_inline operator double() noexcept(false);
+  explicit simdjson_inline operator double() noexcept(false);
   /**
    * Cast this JSON value to a string.
    *
@@ -93329,7 +116324,7 @@ public:
    *          time it parses a document or when it is destroyed.
    * @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not a string.
    */
-  simdjson_inline operator std::string_view() noexcept(false) simdjson_lifetime_bound;
+  explicit simdjson_inline operator std::string_view() noexcept(false) simdjson_lifetime_bound;
   /**
    * Cast this JSON value to a raw_json_string.
    *
@@ -93338,14 +116333,14 @@ public:
    * @returns A pointer to the raw JSON for the given string.
    * @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not a string.
    */
-  simdjson_inline operator raw_json_string() noexcept(false) simdjson_lifetime_bound;
+  explicit simdjson_inline operator raw_json_string() noexcept(false) simdjson_lifetime_bound;
   /**
    * Cast this JSON value to a bool.
    *
    * @returns A bool value.
    * @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not true or false.
    */
-  simdjson_inline operator bool() noexcept(false);
+  explicit simdjson_inline operator bool() noexcept(false);
   /**
    * Cast this JSON value to a value when the document is an object or an array.
    *
@@ -93840,9 +116835,24 @@ public:
   simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template <typename string_type>
   simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -93854,7 +116864,7 @@ public:
   template <typename T>
   simdjson_inline simdjson_result<T> get() &
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    noexcept(nothrow_gettable<T, document_reference>)
 #else
     noexcept
 #endif
@@ -93867,7 +116877,8 @@ public:
   template<typename T>
   simdjson_inline simdjson_result<T> get() &&
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    // Forwards to document::get<T>(), so the document customization decides.
+    noexcept(nothrow_gettable<T, document>)
 #else
     noexcept
 #endif
@@ -93879,7 +116890,8 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool, value
+   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool, value
    *
    * Be mindful that the document instance must remain in scope while you are accessing object, array and value instances.
    *
@@ -93890,7 +116902,7 @@ public:
   template<typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out) &
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document_reference> : true)
+    noexcept(nothrow_gettable<T, document_reference>)
 #else
     noexcept
 #endif
@@ -93903,7 +116915,7 @@ public:
         "And you do not seem to have added support for it. Indeed, we have that "
         "simdjson::custom_deserializable<T> is false and the type T is not a default type "
         "such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-        "int64_t, double, or bool.");
+        "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
       static_cast<void>(out); // to get rid of unused errors
       return UNINITIALIZED;
     }
@@ -93912,7 +116924,8 @@ public:
     // immediately fail.
     static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
       "The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-      "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+      "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+      "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
       " get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
       " You may also add support for custom types, see our documentation.");
     static_cast<void>(out); // to get rid of unused errors
@@ -93921,7 +116934,12 @@ public:
   }

   /** @overload template<typename T> error_code get(T &out) & noexcept */
-  template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+  template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, document_reference>);
+#else
+    noexcept;
+#endif
   simdjson_inline simdjson_result<std::string_view> raw_json() noexcept;
 #if SIMDJSON_STATIC_REFLECTION
   template<constevalutil::fixed_string... FieldNames, typename T>
@@ -93934,12 +116952,12 @@ public:
   explicit simdjson_inline operator T() noexcept(false);
   simdjson_inline operator array() & noexcept(false);
   simdjson_inline operator object() & noexcept(false);
-  simdjson_inline operator uint64_t() noexcept(false);
-  simdjson_inline operator int64_t() noexcept(false);
-  simdjson_inline operator double() noexcept(false);
-  simdjson_inline operator std::string_view() noexcept(false);
-  simdjson_inline operator raw_json_string() noexcept(false);
-  simdjson_inline operator bool() noexcept(false);
+  explicit simdjson_inline operator uint64_t() noexcept(false);
+  explicit simdjson_inline operator int64_t() noexcept(false);
+  explicit simdjson_inline operator double() noexcept(false);
+  explicit simdjson_inline operator std::string_view() noexcept(false);
+  explicit simdjson_inline operator raw_json_string() noexcept(false);
+  explicit simdjson_inline operator bool() noexcept(false);
   simdjson_inline operator value() noexcept(false);
 #endif
   simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -94001,9 +117019,24 @@ public:
   simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template <typename string_type>
   simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -94012,11 +117045,31 @@ public:
   simdjson_inline simdjson_result<haswell::ondemand::value> get_value() noexcept;
   simdjson_inline simdjson_result<bool> is_null() noexcept;

-  template<typename T> simdjson_inline simdjson_result<T> get() & noexcept;
-  template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() && noexcept;
+  template<typename T> simdjson_inline simdjson_result<T> get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, haswell::ondemand::document>);
+#else
+    noexcept;
+#endif
+  template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, haswell::ondemand::document>);
+#else
+    noexcept;
+#endif

-  template<typename T> simdjson_inline error_code get(T &out) & noexcept;
-  template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+  template<typename T> simdjson_inline error_code get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, haswell::ondemand::document>);
+#else
+    noexcept;
+#endif
+  template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, haswell::ondemand::document>);
+#else
+    noexcept;
+#endif
 #if SIMDJSON_EXCEPTIONS

   using haswell::implementation_simdjson_result_base<haswell::ondemand::document>::operator*;
@@ -94025,12 +117078,12 @@ public:
   explicit simdjson_inline operator T() noexcept(false);
   simdjson_inline operator haswell::ondemand::array() & noexcept(false);
   simdjson_inline operator haswell::ondemand::object() & noexcept(false);
-  simdjson_inline operator uint64_t() noexcept(false);
-  simdjson_inline operator int64_t() noexcept(false);
-  simdjson_inline operator double() noexcept(false);
-  simdjson_inline operator std::string_view() noexcept(false);
-  simdjson_inline operator haswell::ondemand::raw_json_string() noexcept(false);
-  simdjson_inline operator bool() noexcept(false);
+  explicit simdjson_inline operator uint64_t() noexcept(false);
+  explicit simdjson_inline operator int64_t() noexcept(false);
+  explicit simdjson_inline operator double() noexcept(false);
+  explicit simdjson_inline operator std::string_view() noexcept(false);
+  explicit simdjson_inline operator haswell::ondemand::raw_json_string() noexcept(false);
+  explicit simdjson_inline operator bool() noexcept(false);
   simdjson_inline operator haswell::ondemand::value() noexcept(false);
 #endif
   simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -94096,9 +117149,24 @@ public:
   simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template <typename string_type>
   simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -94107,22 +117175,42 @@ public:
   simdjson_inline simdjson_result<haswell::ondemand::value> get_value() noexcept;
   simdjson_inline simdjson_result<bool> is_null() noexcept;

-  template<typename T> simdjson_inline simdjson_result<T> get() & noexcept;
-  template<typename T> simdjson_inline simdjson_result<T> get() && noexcept;
+  template<typename T> simdjson_inline simdjson_result<T> get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, haswell::ondemand::document_reference>);
+#else
+    noexcept;
+#endif
+  template<typename T> simdjson_inline simdjson_result<T> get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, haswell::ondemand::document_reference>);
+#else
+    noexcept;
+#endif

-  template<typename T> simdjson_inline error_code get(T &out) & noexcept;
-  template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+  template<typename T> simdjson_inline error_code get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, haswell::ondemand::document_reference>);
+#else
+    noexcept;
+#endif
+  template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, haswell::ondemand::document_reference>);
+#else
+    noexcept;
+#endif
 #if SIMDJSON_EXCEPTIONS
   template <class T>
   explicit simdjson_inline operator T() noexcept(false);
   simdjson_inline operator haswell::ondemand::array() & noexcept(false);
   simdjson_inline operator haswell::ondemand::object() & noexcept(false);
-  simdjson_inline operator uint64_t() noexcept(false);
-  simdjson_inline operator int64_t() noexcept(false);
-  simdjson_inline operator double() noexcept(false);
-  simdjson_inline operator std::string_view() noexcept(false);
-  simdjson_inline operator haswell::ondemand::raw_json_string() noexcept(false);
-  simdjson_inline operator bool() noexcept(false);
+  explicit simdjson_inline operator uint64_t() noexcept(false);
+  explicit simdjson_inline operator int64_t() noexcept(false);
+  explicit simdjson_inline operator double() noexcept(false);
+  explicit simdjson_inline operator std::string_view() noexcept(false);
+  explicit simdjson_inline operator haswell::ondemand::raw_json_string() noexcept(false);
+  explicit simdjson_inline operator bool() noexcept(false);
   simdjson_inline operator haswell::ondemand::value() noexcept(false);
 #endif
   simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -94290,10 +117378,7 @@ public:
    *   }
    *   size_t truncated = stream.truncated_bytes();
    *
-   * IMPORTANT: this value is only meaningful under the conditions below. It is
-   * computed from stage-1 bookkeeping, and outside these conditions it is not
-   * merely imprecise, it is arbitrary -- it can exceed size_in_bytes() or wrap
-   * around to a huge value. Check it only when both of the following hold:
+   * IMPORTANT: this value is only meaningful under the conditions below.
    *
    *   - you iterated all the way to the end of the stream;
    *   - no document reported an error. Iteration stops at the first failed
@@ -94302,6 +117387,9 @@ public:
    * If you need to know about a truncated tail outside those conditions, track
    * it yourself from the last successful document (see iterator::current_index()
    * and iterator::source()).
+   *
+   * An empty input (zero bytes) or an input made only of white space contains
+   * no document: truncated_bytes() returns zero.
    */
   inline size_t truncated_bytes() const noexcept;

@@ -94361,7 +117449,10 @@ public:
      *
      * The returned string_view instance is simply a map to the (unparsed)
      * source string: it may thus include white-space characters and all manner
-     * of padding.
+     * of padding. It spans the whole current document, whether or not you
+     * have already accessed (part of) the document. Thus
+     * current_index() + source().size() is the offset just past the end of the
+     * current document, which is useful when reading a stream in chunks.
      *
      * This function (source()) is experimental and the usage
      * may change in future versions of simdjson: we find the API somewhat
@@ -94415,13 +117506,16 @@ private:
    * @param buf is the raw byte buffer we need to process
    * @param len is the length of the raw byte buffer in bytes
    * @param batch_size is the size of the windows (must be strictly greater or equal to the largest JSON document)
+   * @param allow_comma_separated whether to allow comma-separated documents
+   * @param format the stream format
    */
   simdjson_inline document_stream(
     ondemand::parser &parser,
     const uint8_t *buf,
     size_t len,
     size_t batch_size,
-    bool allow_comma_separated
+    bool allow_comma_separated,
+    stream_format format = stream_format::whitespace_delimited
   ) noexcept;

   /**
@@ -94455,8 +117549,23 @@ private:
    */
   inline void next() noexcept;

-  /** Move the json_iterator of the document to the location of the next document in the stream. */
+  /**
+   * Move the json_iterator of the document to the location of the next document
+   * in the stream.
+   *
+   * For formats with a document delimiter (`newline_delimited`, `json_sequence`),
+   * when the iterator is still inside the current document (`depth() > 0`), this
+   * may jump to the next delimiter instead of walking remaining structurals. That
+   * jump does not structure-validate the unread remainder.
+   */
   inline void next_document() noexcept;
+  /** Byte that ends a document under `format`, or 0 if there is none. */
+  simdjson_inline uint8_t document_delimiter() const noexcept;
+  /**
+   * Position the iterator at the first structural at or past the next
+   * `delimiter` in the current batch. Returns false if none is found.
+   */
+  simdjson_inline bool skip_to_delimiter(uint8_t delimiter) noexcept;

   /** Get the next document index. */
   inline size_t next_batch_start() const noexcept;
@@ -94470,6 +117579,7 @@ private:
   size_t len;
   size_t batch_size;
   bool allow_comma_separated;
+  stream_format format;
   /**
    * We are going to use just one document instance. The document owns
    * the json_iterator. It implies that we only ever pass a reference
@@ -94496,7 +117606,7 @@ private:
   /** The error returned from the stage 1 thread. */
   error_code stage1_thread_error{UNINITIALIZED};
   /** The thread used to run stage 1 against the next batch in the background. */
-  std::unique_ptr<stage1_worker> worker{new(std::nothrow) stage1_worker()};
+  std::unique_ptr<stage1_worker> worker{};
   /**
    * The parser used to run stage 1 in the background. Will be swapped
    * with the regular parser when finished.
@@ -94571,6 +117681,16 @@ public:
    * call it again nor can you call key().
    */
   simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  /**
+   * Get the key as a C++20 u8string_view. This is a zero-copy alias of
+   * unescaped_key(): the very same bytes are returned, viewed as char8_t.
+   *
+   * This consumes the key: once you have called unescaped_u8key(), you cannot
+   * call it again nor can you call key().
+   */
+  simdjson_inline simdjson_warn_unused simdjson_result<std::u8string_view> unescaped_u8key(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   /**
    * Get the key as a string_view (for higher speed, consider raw_key).
    * We deliberately use a more cumbersome name (unescaped_key) to force users
@@ -94603,6 +117723,16 @@ public:
    * you can safely call it repeatedly. Use unescaped_key() to get the unescaped key.
    */
   simdjson_inline std::string_view escaped_key() const noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  /**
+   * Get the key as a C++20 u8string_view. This is a zero-copy alias of
+   * escaped_key(): the very same bytes are returned, viewed as char8_t.
+   * The string is unprocessed, so it may contain escape characters
+   * (e.g., \uXXXX or \n). It does not count as a consumption of the content:
+   * you can safely call it repeatedly.
+   */
+  simdjson_inline std::u8string_view escaped_u8key() const noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   /**
    * Get the field value.
    */
@@ -94634,11 +117764,17 @@ public:
   simdjson_inline simdjson_result() noexcept = default;

   simdjson_inline simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> unescaped_u8key(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template<typename string_type>
   simdjson_inline error_code unescaped_key(string_type &receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<haswell::ondemand::raw_json_string> key() noexcept;
   simdjson_inline simdjson_result<std::string_view> key_raw_json_token() noexcept;
   simdjson_inline simdjson_result<std::string_view> escaped_key() noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> escaped_u8key() noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   simdjson_inline simdjson_result<haswell::ondemand::value> value() noexcept;
 };

@@ -94646,6 +117782,1398 @@ public:

 #endif // SIMDJSON_GENERIC_ONDEMAND_FIELD_H
 /* end file simdjson/generic/ondemand/field.h for haswell */
+/* including simdjson/generic/ondemand/key_selector.h for haswell: #include "simdjson/generic/ondemand/key_selector.h" */
+/* begin file simdjson/generic/ondemand/key_selector.h for haswell */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+#define SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #include "simdjson/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/common_defs.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/constevalutil.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/raw_json_string.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#include <array>
+#include <string>      // std::string (key_selector::describe)
+#include <string_view>
+#include <cstddef>
+#include <cstdint>
+#include <cstring>     // std::memcpy (portable unaligned window load)
+#include <utility>     // std::index_sequence (window candidate dispatch)
+#include <type_traits> // std::integral_constant (window candidate dispatch)
+
+// AArch64 only: 32-bit ARM (ARMv7) defines __ARM_NEON too, but lacks the
+// A64-only horizontal reductions (vmaxvq_u32) used below. See issue #2885.
+#if defined(__aarch64__)
+  #include <arm_neon.h>
+  #define SIMDJSON_KEY_SELECTOR_HAS_NEON 1
+#else
+  #define SIMDJSON_KEY_SELECTOR_HAS_NEON 0
+#endif
+#if defined(__SSE2__)
+  #include <emmintrin.h>
+  #define SIMDJSON_KEY_SELECTOR_HAS_SSE2 1
+#else
+  #define SIMDJSON_KEY_SELECTOR_HAS_SSE2 0
+#endif
+#if defined(__loongarch_sx)
+  #include <lsxintrin.h>
+  #define SIMDJSON_KEY_SELECTOR_HAS_LSX 1
+#else
+  #define SIMDJSON_KEY_SELECTOR_HAS_LSX 0
+#endif
+
+#if SIMDJSON_SUPPORTS_CONCEPTS
+
+namespace simdjson {
+namespace haswell {
+namespace ondemand {
+
+namespace key_selector_detail {
+
+// Since not constexpr, triggers compile-time error.
+inline void compile_time_error(const char* message) noexcept { (void)message; }
+
+// ============================================================================
+// Compile-time perfect-hash generator.
+// It scales to ~100 keys at compile time by determining association values one (position, character)
+// symbol at a time (gperf-style) instead of an exhaustive offset search, and
+// falls back to a Hash-and-Displace construction for large/awkward key sets.
+//
+// Only flat tables survive to runtime; the lookup is a few additions plus a
+// single SIMD key comparison (see match_raw below).
+// ============================================================================
+
+// Maximum number of character positions the gperf hash may combine.
+static constexpr std::size_t MAX_POSITIONS = 16;
+// Sentinel "position" meaning "the last character of the key".
+static constexpr std::size_t LAST_CHAR = std::size_t(-1);
+// Runtime-encoded sentinels (stored in uint8 tables).
+static constexpr std::uint8_t POS_LAST_CHAR = 0xFF; // positions_[i] == last char
+static constexpr std::uint8_t HD_MODE       = 0xFF; // num_positions == H&D mode
+// Flags stored in positions[2] in H&D mode to select the key-hash variant.
+static constexpr std::size_t HD_HASH_2BYTE_FLAG = 2;
+static constexpr std::size_t HD_HASH_4BYTE_FLAG = 4;
+
+constexpr std::size_t next_power_of_2(std::size_t n) noexcept {
+    if (n == 0) { return 1; }
+    std::size_t p = 1;
+    while (p < n) { p <<= 1; }
+    return p;
+}
+
+// Character at a given position (LAST_CHAR means last character), or 256 if out
+// of bounds.
+constexpr std::size_t char_at(std::string_view key, std::size_t pos) noexcept {
+    if (pos == LAST_CHAR) {
+        if (key.empty()) { return 256; }
+        return static_cast<unsigned char>(key[key.size() - 1]);
+    }
+    if (pos >= key.size()) { return 256; }
+    return static_cast<unsigned char>(key[pos]);
+}
+
+// Count key pairs that a set of positions fails to distinguish. Keys whose
+// lengths differ modulo the table size are separated by the length term in the
+// hash, so they need no position coverage.
+template <std::size_t N>
+consteval std::size_t count_undistinguished_pairs(
+    const std::array<std::string_view, N>& keys,
+    const std::size_t* positions,
+    std::size_t num_positions,
+    std::size_t modulus) {
+    std::size_t count = 0;
+    for (std::size_t i = 0; i < N; ++i) {
+        for (std::size_t j = i + 1; j < N; ++j) {
+            if (keys[i].size() % modulus != keys[j].size() % modulus) { continue; }
+            bool distinguished = false;
+            for (std::size_t p = 0; p < num_positions; ++p) {
+                if (char_at(keys[i], positions[p]) != char_at(keys[j], positions[p])) {
+                    distinguished = true;
+                    break;
+                }
+            }
+            if (!distinguished) { ++count; }
+        }
+    }
+    return count;
+}
+
+template <std::size_t N>
+consteval bool positions_distinguish(
+    const std::array<std::string_view, N>& keys,
+    const std::size_t* positions,
+    std::size_t num_positions,
+    std::size_t modulus) {
+    return count_undistinguished_pairs<N>(keys, positions, num_positions, modulus) == 0;
+}
+
+// Number of distinct (length % modulus, char_at(key, pos)) pairs at a position.
+template <std::size_t N>
+consteval std::size_t discriminating_power(
+    const std::array<std::string_view, N>& keys,
+    std::size_t pos,
+    std::size_t modulus) {
+    struct pair { std::size_t len_mod; std::size_t ch; };
+    std::array<pair, N> pairs{};
+    for (std::size_t i = 0; i < N; ++i) {
+        pairs[i] = {keys[i].size() % modulus, char_at(keys[i], pos)};
+    }
+    std::size_t count = 0;
+    for (std::size_t i = 0; i < N; ++i) {
+        bool dup = false;
+        for (std::size_t j = 0; j < i; ++j) {
+            if (pairs[i].len_mod == pairs[j].len_mod && pairs[i].ch == pairs[j].ch) {
+                dup = true;
+                break;
+            }
+        }
+        if (!dup) { ++count; }
+    }
+    return count;
+}
+
+template <std::size_t N>
+consteval std::size_t max_key_length(const std::array<std::string_view, N>& keys) {
+    std::size_t m = 0;
+    for (std::size_t i = 0; i < N; ++i) {
+        if (keys[i].size() > m) { m = keys[i].size(); }
+    }
+    return m;
+}
+
+// Bounded backtracking DFS for a minimal set of distinguishing positions.
+template <std::size_t N>
+consteval bool backtracking_search(
+    const std::array<std::string_view, N>& keys,
+    const std::size_t* candidates,
+    std::size_t num_candidates,
+    std::size_t* positions,
+    std::size_t& num_positions_out,
+    std::size_t& budget,
+    std::size_t modulus) {
+    constexpr std::size_t MAX_DEPTH = 8;
+    std::size_t breadth = num_candidates < 20 ? num_candidates : 20;
+
+    struct frame { std::size_t depth; std::size_t next_ci; std::size_t parent_count; };
+    std::array<frame, MAX_DEPTH + 1> stack{};
+    std::size_t sp = 0;
+
+    std::size_t initial_count = count_undistinguished_pairs<N>(keys, positions, 0, modulus);
+    if (budget > 0) { --budget; }
+    if (initial_count == 0) { num_positions_out = 0; return true; }
+
+    stack[0] = {0, 0, initial_count};
+
+    while (budget > 0) {
+        if (sp > MAX_DEPTH) {
+            if (sp == 0) { break; }
+            --sp;
+            ++stack[sp].next_ci;
+            continue;
+        }
+        auto& f = stack[sp];
+        if (f.next_ci >= breadth) {
+            if (sp == 0) { break; }
+            --sp;
+            ++stack[sp].next_ci;
+            continue;
+        }
+        positions[sp] = candidates[f.next_ci];
+        --budget;
+        std::size_t new_count = count_undistinguished_pairs<N>(keys, positions, sp + 1, modulus);
+        if (new_count == 0) { num_positions_out = sp + 1; return true; }
+        if (new_count < f.parent_count && sp + 1 < MAX_DEPTH) {
+            stack[sp + 1] = {sp + 1, f.next_ci + 1, new_count};
+            ++sp;
+        } else {
+            ++f.next_ci;
+        }
+    }
+    return false;
+}
+
+// Phase 1: select character positions that distinguish all colliding pairs.
+template <std::size_t N>
+consteval std::size_t select_positions(
+    const std::array<std::string_view, N>& keys,
+    std::array<std::size_t, MAX_POSITIONS>& positions,
+    std::size_t modulus) {
+    if (positions_distinguish<N>(keys, positions.data(), 0, modulus)) { return 0; }
+
+    std::size_t max_len = max_key_length(keys);
+    constexpr std::size_t MAX_CANDIDATES = 256;
+    std::array<std::size_t, MAX_CANDIDATES> candidates{};
+    std::array<std::size_t, MAX_CANDIDATES> powers{};
+    std::size_t num_candidates = 0;
+    for (std::size_t p = 0; p < max_len && num_candidates < MAX_CANDIDATES - 1; ++p) {
+        candidates[num_candidates] = p;
+        powers[num_candidates] = discriminating_power(keys, p, modulus);
+        ++num_candidates;
+    }
+    if (num_candidates < MAX_CANDIDATES) {
+        candidates[num_candidates] = LAST_CHAR;
+        powers[num_candidates] = discriminating_power(keys, LAST_CHAR, modulus);
+        ++num_candidates;
+    }
+    for (std::size_t i = 0; i < num_candidates; ++i) {
+        for (std::size_t j = i + 1; j < num_candidates; ++j) {
+            if (powers[j] > powers[i]) {
+                auto tc = candidates[i]; candidates[i] = candidates[j]; candidates[j] = tc;
+                auto tp = powers[i]; powers[i] = powers[j]; powers[j] = tp;
+            }
+        }
+    }
+
+    positions[0] = candidates[0];
+    if (positions_distinguish<N>(keys, positions.data(), 1, modulus)) { return 1; }
+
+    positions[0] = 0;
+    positions[1] = LAST_CHAR;
+    if (positions_distinguish<N>(keys, positions.data(), 2, modulus)) { return 2; }
+
+    {
+        std::size_t budget = 5000;
+        std::size_t num_found = 0;
+        if (backtracking_search<N>(keys, candidates.data(), num_candidates,
+                                   positions.data(), num_found, budget, modulus)) {
+            return num_found;
+        }
+    }
+
+    std::size_t num_pos = 0;
+    for (std::size_t ci = 0; ci < num_candidates && num_pos < MAX_POSITIONS; ++ci) {
+        bool already = false;
+        for (std::size_t p = 0; p < num_pos; ++p) {
+            if (positions[p] == candidates[ci]) { already = true; break; }
+        }
+        if (already) { continue; }
+        positions[num_pos] = candidates[ci];
+        ++num_pos;
+        if (positions_distinguish<N>(keys, positions.data(), num_pos, modulus)) { return num_pos; }
+    }
+
+    compile_time_error("Failed to find distinguishing positions for perfect hash");
+    return 0;
+}
+
+// Result of PHF computation. A max-sized slot_to_key array lets the same struct
+// type carry any chosen table size.
+template <std::size_t N>
+struct phf_result {
+    // Allow up to 8x the minimum table size. Sparser tables solve faster.
+    static constexpr std::size_t MAX_TABLE_SIZE = next_power_of_2(N) * 8;
+    std::size_t table_size{};
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso_values{};
+    std::size_t num_positions{};
+    std::array<std::size_t, MAX_POSITIONS> positions{};
+    std::array<std::size_t, MAX_TABLE_SIZE> slot_to_key{};
+};
+
+// Partition-based asso_values search (gperf-style). Determines asso_values one
+// (position, character) symbol at a time; never revisits a value. Equivalence
+// classes (keys sharing the same undetermined symbols) keep the search cheap.
+template <std::size_t N, std::size_t M>
+consteval bool try_generate_gperf(
+    const std::array<std::string_view, N>& keys,
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS>& asso_values,
+    std::size_t& num_positions,
+    std::array<std::size_t, MAX_POSITIONS>& positions,
+    std::array<std::size_t, M>& slot_to_key) {
+    num_positions = select_positions<N>(keys, positions, M);
+
+    for (std::size_t p = 0; p < MAX_POSITIONS; ++p) {
+        for (std::size_t c = 0; c < 256; ++c) { asso_values[p][c] = 0; }
+    }
+
+    if (num_positions == 0) {
+        for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+        for (std::size_t i = 0; i < N; ++i) {
+            std::size_t slot = keys[i].size() % M;
+            if (slot_to_key[slot] != N) { return false; }
+            slot_to_key[slot] = i;
+        }
+        return true;
+    }
+
+    std::array<std::array<std::size_t, MAX_POSITIONS>, N> kchars{};
+    for (std::size_t k = 0; k < N; ++k) {
+        for (std::size_t p = 0; p < num_positions; ++p) {
+            kchars[k][p] = char_at(keys[k], positions[p]);
+        }
+    }
+
+    struct sym_t { std::size_t pos; std::size_t ch; std::size_t freq; };
+    constexpr std::size_t MAX_SYMS = MAX_POSITIONS * 256;
+    std::array<sym_t, MAX_SYMS> syms{};
+    std::size_t nsyms = 0;
+    for (std::size_t p = 0; p < num_positions; ++p) {
+        std::array<std::size_t, 256> freq{};
+        for (std::size_t k = 0; k < N; ++k) {
+            std::size_t c = kchars[k][p];
+            if (c < 256) { freq[c]++; }
+        }
+        for (std::size_t c = 0; c < 256; ++c) {
+            if (freq[c] > 0) { syms[nsyms++] = {p, c, freq[c]}; }
+        }
+    }
+    for (std::size_t i = 0; i < nsyms; ++i) {
+        for (std::size_t j = i + 1; j < nsyms; ++j) {
+            if (syms[j].freq > syms[i].freq) {
+                auto tmp = syms[i]; syms[i] = syms[j]; syms[j] = tmp;
+            }
+        }
+    }
+
+    std::array<std::size_t, N> phash{};
+    for (std::size_t k = 0; k < N; ++k) { phash[k] = keys[k].size(); }
+
+    std::array<std::array<uint64_t, 256>, MAX_POSITIONS> salt{};
+    {
+        uint64_t s = 0x9e3779b97f4a7c15ULL;
+        for (std::size_t p = 0; p < num_positions; ++p) {
+            for (std::size_t c = 0; c < 256; ++c) {
+                s = s * 6364136223846793005ULL + 1442695040888963407ULL;
+                salt[p][c] = s;
+            }
+        }
+    }
+    std::array<uint64_t, N> sig{};
+    for (std::size_t k = 0; k < N; ++k) {
+        uint64_t s = 0;
+        for (std::size_t p = 0; p < num_positions; ++p) {
+            std::size_t c = kchars[k][p];
+            if (c < 256) { s ^= salt[p][c]; }
+        }
+        sig[k] = s;
+    }
+    std::array<std::size_t, N> order{};
+    for (std::size_t k = 0; k < N; ++k) { order[k] = k; }
+
+    std::array<std::size_t, M> slot_gen{};
+    std::size_t gen = 0;
+
+    std::size_t search_limit = next_power_of_2(M);
+    if (search_limit < 32) { search_limit = 32; }
+
+    for (std::size_t si = 0; si < nsyms; ++si) {
+        std::size_t sp = syms[si].pos;
+        std::size_t sc = syms[si].ch;
+
+        uint64_t sp_salt = salt[sp][sc];
+        for (std::size_t k = 0; k < N; ++k) {
+            if (kchars[k][sp] == sc) { sig[k] ^= sp_salt; }
+        }
+
+        for (std::size_t i = 1; i < N; ++i) {
+            std::size_t x = order[i];
+            uint64_t xs = sig[x];
+            std::size_t j = i;
+            while (j > 0 && sig[order[j - 1]] > xs) {
+                order[j] = order[j - 1];
+                --j;
+            }
+            order[j] = x;
+        }
+
+        bool found = false;
+        for (std::size_t v = 0; v < search_limit && !found; ++v) {
+            bool collision = false;
+            std::size_t ci = 0;
+            while (ci < N && !collision) {
+                uint64_t class_sig = sig[order[ci]];
+                std::size_t cj = ci;
+                while (cj < N && sig[order[cj]] == class_sig) { ++cj; }
+                if (cj - ci > 1) {
+                    ++gen;
+                    for (std::size_t x = ci; x < cj; ++x) {
+                        std::size_t k = order[x];
+                        std::size_t h = phash[k];
+                        if (kchars[k][sp] == sc) { h += v; }
+                        h %= M;
+                        if (slot_gen[h] == gen) { collision = true; break; }
+                        slot_gen[h] = gen;
+                    }
+                }
+                ci = cj;
+            }
+            if (!collision) {
+                asso_values[sp][sc] = v;
+                for (std::size_t k = 0; k < N; ++k) {
+                    if (kchars[k][sp] == sc) { phash[k] += v; }
+                }
+                found = true;
+            }
+        }
+        if (!found) { return false; }
+    }
+
+    for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+    for (std::size_t i = 0; i < N; ++i) {
+        std::size_t slot = phash[i] % M;
+        if (slot_to_key[slot] != N) { return false; }
+        slot_to_key[slot] = i;
+    }
+    std::size_t filled = 0;
+    for (std::size_t i = 0; i < M; ++i) {
+        if (slot_to_key[i] != N) { ++filled; }
+    }
+    return filled == N;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_compute_phf(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+    static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso{};
+    std::size_t npos{};
+    std::array<std::size_t, MAX_POSITIONS> pos{};
+    std::array<std::size_t, M> s2k{};
+    if (try_generate_gperf<N, M>(keys, asso, npos, pos, s2k)) {
+        result.table_size = M;
+        result.asso_values = asso;
+        result.num_positions = npos;
+        result.positions = pos;
+        for (std::size_t i = 0; i < M; ++i) { result.slot_to_key[i] = s2k[i]; }
+        for (std::size_t i = M; i < phf_result<N>::MAX_TABLE_SIZE; ++i) { result.slot_to_key[i] = N; }
+        return true;
+    }
+    return false;
+}
+
+template <std::size_t N, std::size_t M, std::size_t MaxM>
+consteval bool try_gperf_po2(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+    if (try_compute_phf<N, M>(keys, result)) { return true; }
+    constexpr std::size_t NextM = M * 2;
+    if constexpr (NextM <= MaxM) { return try_gperf_po2<N, NextM, MaxM>(keys, result); }
+    return false;
+}
+
+// --- Hash-and-Displace fallback --------------------------------------------
+
+constexpr std::size_t hd_bucket_hash(std::string_view key) noexcept {
+    std::size_t c0 = key.empty() ? 0 : static_cast<unsigned char>(key[0]);
+    std::size_t c1 = key.empty() ? 0 : static_cast<unsigned char>(key[key.size() - 1]);
+    return (c0 + c1 * 3 + key.size() * 17) & 0xFF;
+}
+constexpr std::size_t hd_safe_char(const char* p, std::size_t len, std::size_t idx) noexcept {
+    std::size_t has = static_cast<std::size_t>(idx < len);
+    std::size_t si = idx & (std::size_t{0} - has);
+    return static_cast<unsigned char>(p[si]) & (std::size_t{0} - has);
+}
+constexpr std::size_t hd_key_hash_2(std::string_view key) noexcept {
+    std::size_t kc = key.size();
+    kc = kc * 31 + static_cast<unsigned char>(key[0]);
+    kc = kc * 31 + hd_safe_char(key.data(), key.size(), 1);
+    return kc;
+}
+constexpr std::size_t hd_key_hash_4(std::string_view key) noexcept {
+    std::size_t kc = key.size();
+    kc = kc * 31 + static_cast<unsigned char>(key[0]);
+    kc = kc * 31 + hd_safe_char(key.data(), key.size(), 1);
+    kc = kc * 31 + hd_safe_char(key.data(), key.size(), 2);
+    kc = kc * 31 + hd_safe_char(key.data(), key.size(), 3);
+    return kc;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_hash_and_displace(
+    const std::array<std::string_view, N>& keys,
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS>& asso_values,
+    std::size_t& num_positions,
+    std::array<std::size_t, MAX_POSITIONS>& positions,
+    std::array<std::size_t, M>& slot_to_key) {
+    num_positions = select_positions<N>(keys, positions, M);
+
+    if (num_positions == 0) {
+        for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+        for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+        for (std::size_t i = 0; i < N; ++i) {
+            std::size_t slot = keys[i].size() % M;
+            if (slot_to_key[slot] != N) { return false; }
+            slot_to_key[slot] = i;
+        }
+        return true;
+    }
+
+    for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+    num_positions = HD_MODE; // sentinel for H&D mode
+    positions[0] = 0;
+    positions[1] = LAST_CHAR;
+
+    std::array<std::size_t, N> key_bucket{};
+    for (std::size_t i = 0; i < N; ++i) { key_bucket[i] = hd_bucket_hash(keys[i]); }
+
+    struct bucket_info { std::size_t ch; std::size_t count; };
+    std::array<bucket_info, N> buckets{};
+    std::size_t num_buckets = 0;
+    for (std::size_t i = 0; i < N; ++i) {
+        std::size_t bk = key_bucket[i];
+        bool found = false;
+        for (std::size_t b = 0; b < num_buckets; ++b) {
+            if (buckets[b].ch == bk) { ++buckets[b].count; found = true; break; }
+        }
+        if (!found) { buckets[num_buckets++] = {bk, 1}; }
+    }
+    for (std::size_t i = 0; i < num_buckets; ++i) {
+        for (std::size_t j = i + 1; j < num_buckets; ++j) {
+            if (buckets[j].count > buckets[i].count) {
+                auto tmp = buckets[i]; buckets[i] = buckets[j]; buckets[j] = tmp;
+            }
+        }
+    }
+
+    auto try_placement = [&](auto key_hash_fn) -> bool {
+        for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+        for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+        for (std::size_t b = 0; b < num_buckets; ++b) {
+            std::size_t ch = buckets[b].ch;
+            std::array<std::size_t, N> bucket_keys{};
+            std::size_t bk_count = 0;
+            for (std::size_t i = 0; i < N; ++i) {
+                if (key_bucket[i] == ch) { bucket_keys[bk_count++] = i; }
+            }
+            bool placed = false;
+            std::size_t max_d = M < 255 ? M : 255;
+            for (std::size_t d = 0; d < max_d; ++d) {
+                bool ok = true;
+                std::array<std::size_t, N> bucket_slots{};
+                for (std::size_t k = 0; k < bk_count; ++k) {
+                    std::size_t slot = (d + key_hash_fn(keys[bucket_keys[k]])) % M;
+                    if (slot_to_key[slot] != N) { ok = false; break; }
+                    for (std::size_t k2 = 0; k2 < k; ++k2) {
+                        if (bucket_slots[k2] == slot) { ok = false; break; }
+                    }
+                    if (!ok) { break; }
+                    bucket_slots[k] = slot;
+                }
+                if (ok) {
+                    asso_values[0][ch] = d;
+                    for (std::size_t k = 0; k < bk_count; ++k) {
+                        slot_to_key[bucket_slots[k]] = bucket_keys[k];
+                    }
+                    placed = true;
+                    break;
+                }
+            }
+            if (!placed) { return false; }
+        }
+        std::size_t filled = 0;
+        for (std::size_t i = 0; i < M; ++i) {
+            if (slot_to_key[i] != N) { ++filled; }
+        }
+        return filled == N;
+    };
+
+    if (try_placement([](std::string_view k) { return hd_key_hash_2(k); })) {
+        positions[2] = HD_HASH_2BYTE_FLAG;
+        return true;
+    }
+    if (try_placement([](std::string_view k) { return hd_key_hash_4(k); })) {
+        positions[2] = HD_HASH_4BYTE_FLAG;
+        return true;
+    }
+    return false;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_compute_phf_hd(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+    static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso{};
+    std::size_t npos{};
+    std::array<std::size_t, MAX_POSITIONS> pos{};
+    std::array<std::size_t, M> s2k{};
+    if (try_hash_and_displace<N, M>(keys, asso, npos, pos, s2k)) {
+        result.table_size = M;
+        result.asso_values = asso;
+        result.num_positions = npos;
+        result.positions = pos;
+        for (std::size_t i = 0; i < M; ++i) { result.slot_to_key[i] = s2k[i]; }
+        for (std::size_t i = M; i < phf_result<N>::MAX_TABLE_SIZE; ++i) { result.slot_to_key[i] = N; }
+        return true;
+    }
+    return false;
+}
+
+template <std::size_t N, std::size_t M>
+consteval phf_result<N> compute_phf_hd_po2(const std::array<std::string_view, N>& keys) {
+    static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+    phf_result<N> result{};
+    if (try_compute_phf_hd<N, M>(keys, result)) { return result; }
+    constexpr std::size_t NextM = M * 2;
+    if constexpr (NextM <= phf_result<N>::MAX_TABLE_SIZE) {
+        return compute_phf_hd_po2<N, NextM>(keys);
+    } else {
+        compile_time_error("Hash-and-Displace: failed to find valid table size");
+        return result;
+    }
+}
+
+// Compute a perfect hash for `keys`: try gperf at power-of-two sizes (capped so
+// the runtime tables stay within uint8 indices), then fall back to H&D.
+template <std::size_t N>
+consteval phf_result<N> compute_phf(const std::array<std::string_view, N>& keys) {
+    constexpr std::size_t StartM = next_power_of_2(N);
+    constexpr std::size_t GPERF_MAX_TABLE =
+        phf_result<N>::MAX_TABLE_SIZE < 256 ? phf_result<N>::MAX_TABLE_SIZE : 256;
+    if constexpr (StartM <= GPERF_MAX_TABLE) {
+        phf_result<N> result{};
+        if (try_gperf_po2<N, StartM, GPERF_MAX_TABLE>(keys, result)) { return result; }
+    }
+    return compute_phf_hd_po2<N, StartM>(keys);
+}
+
+// ============================================================================
+// Runtime tables (flat, uint8) derived from a phf_result.
+// ============================================================================
+
+template <std::size_t N, std::size_t TableSize, std::size_t MaxKeyLen>
+struct phf_data {
+    std::array<std::array<std::uint8_t, 256>, MAX_POSITIONS> asso_values{};
+    std::array<std::uint8_t, MAX_POSITIONS>                  positions{};
+    std::uint8_t                                             num_positions{};
+    std::uint8_t                                             hd_hash_variant{}; // 2 or 4 (H&D only)
+    std::array<std::uint8_t, TableSize>                      slot_to_key{};
+    // slot_key_bytes[s] holds the key stored at slot s, zero-padded to a 16-byte
+    // multiple so the SIMD comparison can read a whole register.
+    std::array<std::array<char, ((MaxKeyLen + 15) / 16) * 16>, TableSize> slot_key_bytes{};
+    std::array<std::uint8_t, TableSize>                      slot_key_len{};
+};
+
+template <std::size_t N>
+constexpr std::size_t compute_max_key_len(const std::array<std::string_view, N>& keys) noexcept {
+    std::size_t m = 0;
+    for (std::size_t i = 0; i < N; ++i) { if (keys[i].size() > m) { m = keys[i].size(); } }
+    return m;
+}
+
+// Validate keys and build the runtime tables from the computed perfect hash.
+template <std::size_t N, std::size_t TableSize, std::size_t MaxKeyLen>
+consteval phf_data<N, TableSize, MaxKeyLen>
+build_phf_data(const std::array<std::string_view, N>& keys, const phf_result<N>& result) {
+    for (std::size_t i = 0; i < N; ++i) {
+        if (keys[i].empty())            { compile_time_error("empty keys are not allowed in key_selector"); }
+        if (keys[i].size() > MaxKeyLen) { compile_time_error("key length exceeds MaxKeyLen"); }
+        for (char c : keys[i]) {
+            if (c == '\\') { compile_time_error("backslash not allowed in key_selector keys"); }
+            if (c == '"')  { compile_time_error("quote not allowed in key_selector keys"); }
+            if (c == '\0') { compile_time_error("null byte not allowed in key_selector keys"); }
+        }
+        for (std::size_t j = i + 1; j < N; ++j) {
+            if (keys[i] == keys[j]) { compile_time_error("duplicate keys in key_selector"); }
+        }
+    }
+
+    phf_data<N, TableSize, MaxKeyLen> out{};
+
+    if (result.num_positions == HD_MODE) {
+        // H&D mode: single displacement table in asso_values[0].
+        for (std::size_t c = 0; c < 256; ++c) {
+            out.asso_values[0][c] = static_cast<std::uint8_t>(result.asso_values[0][c]);
+        }
+        out.num_positions   = static_cast<std::uint8_t>(HD_MODE);
+        out.hd_hash_variant = static_cast<std::uint8_t>(result.positions[2]);
+    } else {
+        for (std::size_t pi = 0; pi < result.num_positions; ++pi) {
+            for (std::size_t c = 0; c < 256; ++c) {
+                out.asso_values[pi][c] = static_cast<std::uint8_t>(result.asso_values[pi][c] % TableSize);
+            }
+        }
+        out.num_positions = static_cast<std::uint8_t>(result.num_positions);
+        for (std::size_t i = 0; i < result.num_positions; ++i) {
+            out.positions[i] = (result.positions[i] == LAST_CHAR)
+                ? POS_LAST_CHAR
+                : static_cast<std::uint8_t>(result.positions[i]);
+        }
+    }
+
+    for (std::size_t s = 0; s < TableSize; ++s) {
+        out.slot_to_key[s] = static_cast<std::uint8_t>(result.slot_to_key[s]);
+    }
+
+    for (std::size_t s = 0; s < TableSize; ++s) {
+        std::size_t ki = result.slot_to_key[s];
+        if (ki < N) {
+            auto k = keys[ki];
+            out.slot_key_len[s] = static_cast<std::uint8_t>(k.size());
+            for (std::size_t c = 0; c < k.size(); ++c) { out.slot_key_bytes[s][c] = k[c]; }
+        } else {
+            out.slot_key_len[s] = 0; // empty slot: no length can match
+        }
+    }
+    return out;
+}
+
+// --- SIMD runtime primitives ------------------------------------------------
+
+// The runtime matchers read whole 16-byte blocks up to offset 63 (four blocks)
+// for keys as long as the 63-character maximum. That read must stay within the
+// buffer's trailing padding, so the padding has to exceed the largest offset we
+// touch.
+static_assert(SIMDJSON_PADDING > 63,
+              "key_selector requires SIMDJSON_PADDING > 63 for its SIMD key reads");
+
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+// True when every byte of `diff` is zero. A 32-bit-lane horizontal max is
+// enough (zero iff every 32-bit word is zero) and is cheaper than a byte-wide
+// reduction. Do not replace this with a floating-point compare against 0.0:
+// flush-to-zero would treat a denormal as zero.
+simdjson_really_inline bool neon_all_bytes_zero(uint8x16_t diff) noexcept {
+    return vmaxvq_u32(vreinterpretq_u32_u8(diff)) == 0;
+}
+#endif
+
+// Scan for the terminating '"' starting at p. Returns its byte offset (= key
+// length). Caller guarantees SIMDJSON_PADDING (== 64) bytes past the JSON buffer,
+// so reading whole 16-byte blocks up to offset 63 is always safe.
+template <std::size_t MaxKeyLen>
+simdjson_really_inline std::size_t scan_key_length(const char* p) noexcept {
+    // The closing quote of the longest key sits at most at offset MaxKeyLen, so we
+    // scan as many 16-byte blocks as it takes to cover offsets 0..MaxKeyLen. The
+    // cap (63) keeps every covered offset within the 64-byte padding guarantee, so
+    // the SIMD and scalar builds agree.
+    static_assert(MaxKeyLen <= 63, "MaxKeyLen must be <= 63 for current SIMD implementations");
+    // [[maybe_unused]]: the scalar fallback (no NEON/SSE2/LSX) does not use it.
+    [[maybe_unused]] constexpr std::size_t num_blocks = MaxKeyLen / 16 + 1;
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+    for (std::size_t b = 0; b < num_blocks; ++b) {
+        uint8x16_t v = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + b * 16);
+        uint8x16_t cmp = vceqq_u8(v, vdupq_n_u8('"'));
+        uint64_t m = vget_lane_u64(
+            vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(cmp), 4)), 0);
+        if (simdjson_likely(m != 0)) { return b * 16 + (std::size_t(__builtin_ctzll(m)) >> 2); }
+    }
+    return MaxKeyLen + 1;
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+    for (std::size_t b = 0; b < num_blocks; ++b) {
+        __m128i v   = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + b * 16));
+        __m128i cmp = _mm_cmpeq_epi8(v, _mm_set1_epi8('"'));
+        unsigned m  = static_cast<unsigned>(_mm_movemask_epi8(cmp));
+        if (simdjson_likely(m != 0)) { return b * 16 + std::size_t(__builtin_ctz(m)); }
+    }
+    return MaxKeyLen + 1;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+    for (std::size_t b = 0; b < num_blocks; ++b) {
+        __m128i v   = __lsx_vld(reinterpret_cast<const void*>(p + b * 16), 0);
+        __m128i cmp = __lsx_vseq_b(v, __lsx_vreplgr2vr_b('"'));
+        // vmskltz_b gathers the per-byte sign bits (set where the byte equals '"')
+        // into the low 16 bits of lane 0, the LSX equivalent of movemask.
+        unsigned m  = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(cmp), 0)) & 0xFFFFu;
+        if (simdjson_likely(m != 0)) { return b * 16 + std::size_t(__builtin_ctz(m)); }
+    }
+    return MaxKeyLen + 1;
+#else
+    for (std::size_t i = 0; i <= MaxKeyLen; ++i)
+        if (p[i] == '"') return i;
+    return MaxKeyLen + 1;
+#endif
+}
+
+// Byte-equal of p[0..len) against stored[0..len). stored is zero-padded past
+// `len`. Input is read over 16, 32, 48, or 64 bytes (padded JSON buffer
+// guaranteed; SIMDJSON_PADDING == 64).
+template <std::size_t MaxKeyLen>
+simdjson_really_inline bool compare_key_bytes(
+    const char* p, const char* stored, std::size_t len) noexcept {
+    // [[maybe_unused]]: only the NEON/SSE2/LSX branches read these; the scalar
+    // fallback build (no SIMD) leaves them unused, which is an error under -Werror.
+    [[maybe_unused]] alignas(16) static constexpr uint8_t idx16[16] =
+        {0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15};
+    if constexpr (MaxKeyLen <= 16) {
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+        uint8x16_t vp = vld1q_u8(reinterpret_cast<const uint8_t*>(p));
+        uint8x16_t vs = vld1q_u8(reinterpret_cast<const uint8_t*>(stored));
+        uint8x16_t mask = vcltq_u8(vld1q_u8(idx16), vdupq_n_u8(static_cast<uint8_t>(len)));
+        uint8x16_t diff = veorq_u8(vandq_u8(vp, mask), vs);
+        return neon_all_bytes_zero(diff);
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+        __m128i vp = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p));
+        __m128i vs = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored));
+        __m128i idx = _mm_load_si128(reinterpret_cast<const __m128i*>(idx16));
+        __m128i mask = _mm_cmplt_epi8(idx, _mm_set1_epi8(static_cast<char>(len)));
+        __m128i eq = _mm_cmpeq_epi8(_mm_and_si128(vp, mask), vs);
+        return _mm_movemask_epi8(eq) == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+        __m128i vp = __lsx_vld(reinterpret_cast<const void*>(p), 0);
+        __m128i vs = __lsx_vld(reinterpret_cast<const void*>(stored), 0);
+        __m128i idx = __lsx_vld(reinterpret_cast<const void*>(idx16), 0);
+        __m128i mask = __lsx_vslt_b(idx, __lsx_vreplgr2vr_b(static_cast<int>(len)));
+        __m128i eq = __lsx_vseq_b(__lsx_vand_v(vp, mask), vs);
+        return (static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq), 0)) & 0xFFFFu) == 0xFFFFu;
+#else
+        for (std::size_t i = 0; i < len; ++i)
+            if (p[i] != stored[i]) return false;
+        return true;
+#endif
+    } else if constexpr (MaxKeyLen <= 32) {
+        [[maybe_unused]] alignas(16) static constexpr uint8_t idx32_hi[16] =
+            {16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31};
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+        uint8x16_t vp_lo = vld1q_u8(reinterpret_cast<const uint8_t*>(p));
+        uint8x16_t vp_hi = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + 16);
+        uint8x16_t vs_lo = vld1q_u8(reinterpret_cast<const uint8_t*>(stored));
+        uint8x16_t vs_hi = vld1q_u8(reinterpret_cast<const uint8_t*>(stored) + 16);
+        uint8x16_t lenv = vdupq_n_u8(static_cast<uint8_t>(len));
+        uint8x16_t m_lo = vcltq_u8(vld1q_u8(idx16),    lenv);
+        uint8x16_t m_hi = vcltq_u8(vld1q_u8(idx32_hi), lenv);
+        uint8x16_t d_lo = veorq_u8(vandq_u8(vp_lo, m_lo), vs_lo);
+        uint8x16_t d_hi = veorq_u8(vandq_u8(vp_hi, m_hi), vs_hi);
+        return neon_all_bytes_zero(vorrq_u8(d_lo, d_hi));
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+        __m128i vp_lo = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p));
+        __m128i vp_hi = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + 16));
+        __m128i vs_lo = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored));
+        __m128i vs_hi = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored + 16));
+        __m128i lenv = _mm_set1_epi8(static_cast<char>(len));
+        __m128i m_lo = _mm_cmplt_epi8(_mm_load_si128(reinterpret_cast<const __m128i*>(idx16)),    lenv);
+        __m128i m_hi = _mm_cmplt_epi8(_mm_load_si128(reinterpret_cast<const __m128i*>(idx32_hi)), lenv);
+        __m128i eq_lo = _mm_cmpeq_epi8(_mm_and_si128(vp_lo, m_lo), vs_lo);
+        __m128i eq_hi = _mm_cmpeq_epi8(_mm_and_si128(vp_hi, m_hi), vs_hi);
+        return (_mm_movemask_epi8(eq_lo) & _mm_movemask_epi8(eq_hi)) == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+        __m128i lenv = __lsx_vreplgr2vr_b(static_cast<int>(len));
+        __m128i vp_lo = __lsx_vld(reinterpret_cast<const void*>(p), 0);
+        __m128i vp_hi = __lsx_vld(reinterpret_cast<const void*>(p + 16), 0);
+        __m128i vs_lo = __lsx_vld(reinterpret_cast<const void*>(stored), 0);
+        __m128i vs_hi = __lsx_vld(reinterpret_cast<const void*>(stored + 16), 0);
+        __m128i m_lo = __lsx_vslt_b(__lsx_vld(reinterpret_cast<const void*>(idx16), 0),    lenv);
+        __m128i m_hi = __lsx_vslt_b(__lsx_vld(reinterpret_cast<const void*>(idx32_hi), 0), lenv);
+        __m128i eq_lo = __lsx_vseq_b(__lsx_vand_v(vp_lo, m_lo), vs_lo);
+        __m128i eq_hi = __lsx_vseq_b(__lsx_vand_v(vp_hi, m_hi), vs_hi);
+        unsigned mlo = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq_lo), 0)) & 0xFFFFu;
+        unsigned mhi = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq_hi), 0)) & 0xFFFFu;
+        return (mlo & mhi) == 0xFFFFu;
+#else
+        for (std::size_t i = 0; i < len; ++i)
+            if (p[i] != stored[i]) return false;
+        return true;
+#endif
+    } else if constexpr (MaxKeyLen <= 64) {
+        // 3 or 4 16-byte blocks (33..48 -> 3, 49..64 -> 4). stored is zero-padded
+        // to exactly num_blocks*16 bytes (KEY_STRIDE), so neither load overruns it.
+        // [[maybe_unused]]: the scalar fallback (no NEON/SSE2/LSX) does not use it.
+        [[maybe_unused]] constexpr std::size_t num_blocks = (MaxKeyLen + 15) / 16;
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+        uint8x16_t base = vld1q_u8(idx16);
+        uint8x16_t lenv = vdupq_n_u8(static_cast<uint8_t>(len));
+        uint8x16_t acc  = vdupq_n_u8(0);
+        for (std::size_t b = 0; b < num_blocks; ++b) {
+            uint8x16_t vp   = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + b * 16);
+            uint8x16_t vs   = vld1q_u8(reinterpret_cast<const uint8_t*>(stored) + b * 16);
+            uint8x16_t idxv = vaddq_u8(base, vdupq_n_u8(static_cast<uint8_t>(b * 16)));
+            uint8x16_t mask = vcltq_u8(idxv, lenv);
+            acc = vorrq_u8(acc, veorq_u8(vandq_u8(vp, mask), vs));
+        }
+        return neon_all_bytes_zero(acc);
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+        __m128i base = _mm_load_si128(reinterpret_cast<const __m128i*>(idx16));
+        __m128i lenv = _mm_set1_epi8(static_cast<char>(len));
+        int eq = 0xFFFF;
+        for (std::size_t b = 0; b < num_blocks; ++b) {
+            __m128i vp   = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + b * 16));
+            __m128i vs   = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored + b * 16));
+            __m128i idxv = _mm_add_epi8(base, _mm_set1_epi8(static_cast<char>(b * 16)));
+            __m128i mask = _mm_cmplt_epi8(idxv, lenv);
+            eq &= _mm_movemask_epi8(_mm_cmpeq_epi8(_mm_and_si128(vp, mask), vs));
+        }
+        return eq == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+        __m128i base = __lsx_vld(reinterpret_cast<const void*>(idx16), 0);
+        __m128i lenv = __lsx_vreplgr2vr_b(static_cast<int>(len));
+        unsigned acc = 0xFFFFu;
+        for (std::size_t b = 0; b < num_blocks; ++b) {
+            __m128i vp   = __lsx_vld(reinterpret_cast<const void*>(p + b * 16), 0);
+            __m128i vs   = __lsx_vld(reinterpret_cast<const void*>(stored + b * 16), 0);
+            __m128i idxv = __lsx_vadd_b(base, __lsx_vreplgr2vr_b(static_cast<int>(b * 16)));
+            __m128i mask = __lsx_vslt_b(idxv, lenv);
+            __m128i eq   = __lsx_vseq_b(__lsx_vand_v(vp, mask), vs);
+            acc &= static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq), 0)) & 0xFFFFu;
+        }
+        return acc == 0xFFFFu;
+#else
+        for (std::size_t i = 0; i < len; ++i)
+            if (p[i] != stored[i]) return false;
+        return true;
+#endif
+    } else {
+        for (std::size_t i = 0; i < len; ++i)
+            if (p[i] != stored[i]) return false;
+        return true;
+    }
+}
+
+// --- Single 8-bit window fast path ------------------------------------------
+//
+// Many small key sets can be told apart by inspecting a *single* 8-bit window of
+// the key bytes -- and that window need not be byte-aligned. Because every JSON
+// key is terminated by a '"', the bytes at and before a key's length are well
+// defined for any key at least that long: byte i is the key character when i is
+// inside the key and the closing quote when i == len. So we read two bytes at a
+// fixed offset, extract 8 consecutive bits at a fixed intra-byte shift, and if
+// that value is distinct for every key, a 256-entry table maps it straight to a
+// candidate key. The match then needs no hash and -- in the length-free overload
+// -- no SIMD length scan: load two bytes, shift, mask, index the table, and
+// confirm the candidate with one comparison.
+//
+// Allowing an *unaligned* window (a shift of 1..7) mixes bits from two adjacent
+// bytes and discriminates key sets that no single aligned byte can. For example,
+// the partial_tweets keys {created_at,id,text,in_reply_to_status_id,user,
+// retweet_count,favorite_count} share a colliding byte at every aligned position
+// 0,1,2, yet the 8 bits starting at bit offset 2 are unique across all seven.
+// Simpler cases fall out as the shift==0 special case: {"id","screen_name"}
+// splits on byte 0, {"jo","joe"} on the quote at byte 2.
+//
+// The window is confined to the first (shortest key length + 1) bytes so the
+// two-byte read never crosses a key's closing quote into uncontrolled value
+// bytes; that final byte is the shortest key's quote.
+template <std::size_t N, std::size_t MaxKeyLen>
+struct window_data {
+    static constexpr std::size_t KEY_STRIDE = ((MaxKeyLen + 15) / 16) * 16;
+    bool                                            ok{false};
+    std::uint8_t                                    byte_offset{0}; // first byte of the 2-byte read
+    std::uint8_t                                    shift{0};       // intra-byte bit shift (0..7)
+    std::array<std::uint8_t, 256>                   window_to_key{}; // window byte -> key index, N if none
+    std::array<std::uint8_t, N>                     key_len{};
+    std::array<std::array<char, KEY_STRIDE>, N>     key_bytes{};     // zero-padded
+};
+
+// Byte seen at `idx` for key `i`: a key character when idx is inside the key, or
+// the closing '"' when idx == len (callers keep idx <= every key's length).
+template <std::size_t N>
+constexpr unsigned window_byte_at(const std::array<std::string_view, N>& keys,
+                                  std::size_t i, std::size_t idx) noexcept {
+    if (idx < keys[i].size()) { return static_cast<unsigned char>(keys[i][idx]); }
+    return static_cast<unsigned>('"');
+}
+
+// The 8-bit window value for key `i` at (byte_offset, shift): the little-endian
+// pair (byte[off], byte[off+1]) shifted right by `shift` and truncated. Matches
+// the runtime read exactly.
+template <std::size_t N>
+constexpr unsigned window_value(const std::array<std::string_view, N>& keys,
+                                std::size_t i, std::size_t byte_offset, std::size_t shift) noexcept {
+    unsigned lo = window_byte_at<N>(keys, i, byte_offset);
+    unsigned hi = window_byte_at<N>(keys, i, byte_offset + 1);
+    return ((lo | (hi << 8)) >> shift) & 0xFFu;
+}
+
+template <std::size_t N, std::size_t MaxKeyLen>
+consteval window_data<N, MaxKeyLen>
+compute_window(const std::array<std::string_view, N>& keys) {
+    window_data<N, MaxKeyLen> out{};
+
+    std::size_t min_len = keys[0].size();
+    for (std::size_t i = 1; i < N; ++i) {
+        if (keys[i].size() < min_len) { min_len = keys[i].size(); }
+    }
+
+    // Iterate windows nearest the front first (cheapest to read, smallest shift).
+    for (std::size_t off = 0; off <= min_len; ++off) {
+        for (std::size_t shift = 0; shift < 8; ++shift) {
+            // The read touches byte off, and byte off+1 when shift != 0. Both must
+            // stay within the safe region [0, min_len] (min_len is the shortest
+            // key's quote index). off <= min_len is guaranteed by the loop bound.
+            if (shift != 0 && off + 1 > min_len) { continue; }
+
+            bool distinct = true;
+            for (std::size_t i = 0; i < N && distinct; ++i) {
+                for (std::size_t j = i + 1; j < N; ++j) {
+                    if (window_value<N>(keys, i, off, shift) == window_value<N>(keys, j, off, shift)) {
+                        distinct = false;
+                        break;
+                    }
+                }
+            }
+            if (!distinct) { continue; }
+
+            out.ok          = true;
+            out.byte_offset = static_cast<std::uint8_t>(off);
+            out.shift       = static_cast<std::uint8_t>(shift);
+            for (std::size_t b = 0; b < 256; ++b) { out.window_to_key[b] = static_cast<std::uint8_t>(N); }
+            for (std::size_t i = 0; i < N; ++i) {
+                out.window_to_key[window_value<N>(keys, i, off, shift)] = static_cast<std::uint8_t>(i);
+                out.key_len[i] = static_cast<std::uint8_t>(keys[i].size());
+                for (std::size_t c = 0; c < keys[i].size(); ++c) { out.key_bytes[i][c] = keys[i][c]; }
+            }
+            return out;
+        }
+    }
+    return out; // ok == false: no single 8-bit window distinguishes the keys
+}
+
+// Read the 8-bit window at (byte_offset, shift) from a padded key pointer.
+simdjson_really_inline std::uint8_t read_window(const char* p, std::size_t byte_offset,
+                                                std::size_t shift) noexcept {
+    std::uint16_t w;
+    // Two controlled bytes (within the shortest key + its quote, hence within the
+    // padded buffer). memcpy is the portable little-endian unaligned load.
+    std::memcpy(&w, p + byte_offset, sizeof(w));
+#if defined(__BYTE_ORDER__) && (__BYTE_ORDER__ == __ORDER_BIG_ENDIAN__)
+    w = static_cast<std::uint16_t>((w >> 8) | (w << 8));
+#endif
+    return static_cast<std::uint8_t>((w >> shift) & 0xFFu);
+}
+
+// Verify a window candidate. The window table already mapped the key to the
+// single possible index `ki` (< N); here we confirm it. Folding over 0..N-1
+// turns the runtime `ki` into a compile-time index in the matching arm, so the
+// candidate's length and bytes are constants for compare_key_bytes -- the same
+// specialization the ordered find_field path gets from a CT-length
+// unsafe_is_equal, and what keeps this path competitive. The closing-quote check
+// (p[L] == '"') both confirms the key ends exactly at the candidate's length and
+// rejects a wrong-length key, so this works whether or not the caller knew len.
+template <std::size_t N, std::size_t MaxKeyLen, std::size_t... Is>
+simdjson_really_inline std::size_t
+match_window_candidate(const char* p, std::uint8_t ki,
+                       const window_data<N, MaxKeyLen>& w,
+                       std::index_sequence<Is...>) noexcept {
+  std::size_t result = N;
+  auto try_match = [&](auto Ic) {
+    constexpr std::size_t i = decltype(Ic)::value;
+    if (ki == i && p[w.key_len[i]] == '"' &&
+        key_selector_detail::compare_key_bytes<MaxKeyLen>(
+            p, w.key_bytes[i].data(), w.key_len[i])) {
+      result = i;
+    }
+  };
+  (try_match(std::integral_constant<std::size_t, Is>{}), ...);
+  return result;
+}
+
+// --- describe() string helpers (constexpr; no std::to_string, which is not) ---
+
+// Append the decimal form of v to s.
+SIMDJSON_CONSTEXPR_STRING void append_uint(std::string& s, std::size_t v) {
+    if (v == 0) { s.push_back('0'); return; }
+    char buf[20];
+    std::size_t n = 0;
+    while (v > 0) { buf[n++] = static_cast<char>('0' + (v % 10)); v /= 10; }
+    while (n > 0) { s.push_back(buf[--n]); }
+}
+
+// Append a byte as its decimal value, plus the printable character in quotes
+// when it is in the printable ASCII range (e.g. "110 ('n')").
+SIMDJSON_CONSTEXPR_STRING void append_byte(std::string& s, unsigned b) {
+    append_uint(s, b);
+    if (b >= 0x20 && b < 0x7f) {
+        s += " ('";
+        s.push_back(static_cast<char>(b));
+        s += "')";
+    }
+}
+
+} // namespace key_selector_detail
+
+/**
+ * Stateless, compile-time key selector.
+ *
+ * Usage:
+ *   using sel_t = key_selector<"id", "text", "user">;
+ *   std::size_t i = sel_t::match_raw(raw_key); // returns sel_t::size() on miss
+ *
+ * The perfect hash is built at compile time (gperf-style, with a
+ * Hash-and-Displace fallback) and only flat tables survive to runtime. All
+ * tables are static constexpr, so the lookup fully inlines.
+ *
+ * Limitations:
+ *   - Each key must be at most 63 characters long (and no longer than
+ *     SIMDJSON_PADDING). Longer keys trigger a compile-time error.
+ *   - The number of keys should be moderate. The hard limit is 255 keys;
+ *     compilation time grows with the number of keys, so prefer a few dozen at
+ *     most per selector.
+ *   - Keys must be distinct, non-empty, and free of backslash, double-quote and
+ *     null bytes (matching is done against the raw, unescaped JSON key bytes).
+ */
+template <constevalutil::fixed_string... Keys>
+struct key_selector {
+    static constexpr std::size_t N = sizeof...(Keys);
+    static_assert(N > 0,   "key_selector requires at least one key");
+    static_assert(N <= 255,"key_selector supports at most 255 keys");
+
+    static constexpr std::array<std::string_view, N> keys{ Keys.view()... };
+    static constexpr std::size_t max_key_len = key_selector_detail::compute_max_key_len<N>(keys);
+    static_assert(max_key_len <= SIMDJSON_PADDING,
+                  "key longer than SIMDJSON_PADDING is not supported");
+    // The SIMD key-length scan covers offsets 0..63 (four 16-byte blocks), which
+    // stays within the 64-byte padding guarantee. A 64-character key's closing
+    // quote would land at offset 64 and be missed on NEON/SSE2/LSX while still
+    // matching in scalar builds, so cap at 63 to keep implementations in agreement.
+    static_assert(max_key_len <= 63,
+                  "key_selector keys must be at most 63 characters long");
+
+    static constexpr auto result = key_selector_detail::compute_phf<N>(keys);
+    static constexpr std::size_t table_size = result.table_size;
+
+    static constexpr auto phf =
+        key_selector_detail::build_phf_data<N, table_size, max_key_len>(keys, result);
+
+    // Single 8-bit-window discriminator (when one exists). Detected at compile
+    // time and selected with `if constexpr` below, so the hash path is compiled
+    // out for key sets that qualify, and this is compiled out for those that do
+    // not.
+    static constexpr auto window =
+        key_selector_detail::compute_window<N, max_key_len>(keys);
+
+    static constexpr std::size_t size() noexcept { return N; }
+
+    /**
+     * Look up a JSON key whose length is already known. p must point at the first
+     * key byte (just after the opening quote) in a padded simdjson buffer, and len
+     * must be the number of raw key bytes (the distance to the closing quote).
+     * Returns the selector index in [0, N) on match, or N on miss.
+     *
+     * Prefer this overload when the caller can obtain the key length cheaply (for
+     * example, object::for_each derives it from the structural index rather than
+     * re-scanning for the closing quote).
+     */
+    static simdjson_really_inline std::size_t match_raw(const char* p, std::size_t len) noexcept {
+        if (len == 0 || len > max_key_len) { return N; }
+
+        if constexpr (window.ok) {
+            // One 8-bit window selects the only possible candidate key;
+            // match_window_candidate confirms it (bytes + closing quote). p sits
+            // in a padded buffer and the window stays within the shortest key +
+            // quote, so the two-byte read is always in bounds. len is unused here
+            // because the quote check already pins the key's end.
+            std::uint8_t ki = window.window_to_key[
+                key_selector_detail::read_window(p, window.byte_offset, window.shift)];
+            if (ki >= N) { return N; }
+            return key_selector_detail::match_window_candidate(
+                p, ki, window, std::make_index_sequence<N>{});
+        }
+
+        std::size_t slot;
+        if (phf.num_positions == key_selector_detail::HD_MODE) {
+            // Hash-and-Displace: bucket displacement + per-key hash.
+            std::string_view key(p, len);
+            std::size_t bucket = key_selector_detail::hd_bucket_hash(key);
+            std::size_t kh = (phf.hd_hash_variant == key_selector_detail::HD_HASH_2BYTE_FLAG)
+                ? key_selector_detail::hd_key_hash_2(key)
+                : key_selector_detail::hd_key_hash_4(key);
+            slot = (phf.asso_values[0][bucket] + kh) & (table_size - 1);
+        } else {
+            // gperf: h = len + sum of asso_values over the selected positions.
+            // positions / num_positions / asso_values are compile-time constants,
+            // so this loop fully unrolls. The idx < len guard mirrors the
+            // generator's char_at()-> 256 -> skip behavior for out-of-range
+            // positions (required: arbitrary positions may exceed a key's length).
+            std::size_t h = len;
+            for (std::uint8_t i = 0; i < phf.num_positions; ++i) {
+                std::uint8_t pos = phf.positions[i];
+                std::size_t idx = (pos == key_selector_detail::POS_LAST_CHAR)
+                                  ? (len - std::size_t{1})
+                                  : static_cast<std::size_t>(pos);
+                if (idx < len) {
+                    h += phf.asso_values[i][static_cast<unsigned char>(p[idx])];
+                }
+            }
+            slot = h & (table_size - 1);
+        }
+
+        std::uint8_t ki = phf.slot_to_key[slot];
+        if (ki >= N) { return N; }
+        if (phf.slot_key_len[slot] != len) { return N; }
+        if (!key_selector_detail::compare_key_bytes<max_key_len>(
+                p, phf.slot_key_bytes[slot].data(), len)) { return N; }
+        return ki;
+    }
+
+    /**
+     * Look up a JSON key. rjs must point just after an opening quote in a padded
+     * simdjson buffer. Returns the selector index in [0, N) on match, or N on miss.
+     * The key length is recovered with a SIMD scan for the closing quote; callers
+     * that already know the length should use the (p, len) overload above.
+     */
+    static simdjson_really_inline std::size_t match_raw(raw_json_string rjs) noexcept {
+        const char* p = rjs.raw();
+        if constexpr (window.ok) {
+            // One 8-bit window picks the candidate; verifying the candidate's
+            // bytes and its closing '"' confirms the full key, so the length scan
+            // is unnecessary. The window read is in bounds (padding), and the
+            // candidate length is at most max_key_len.
+            std::uint8_t ki = window.window_to_key[
+                key_selector_detail::read_window(p, window.byte_offset, window.shift)];
+            if (ki >= N) { return N; }
+            return key_selector_detail::match_window_candidate(
+                p, ki, window, std::make_index_sequence<N>{});
+        }
+        return match_raw(p, key_selector_detail::scan_key_length<max_key_len>(p));
+    }
+
+    /** Return the key text at selector index i (i in [0, N)). */
+    static constexpr std::string_view key_at(std::size_t i) noexcept {
+        return keys[i];
+    }
+
+    /**
+     * Return a complete, human-readable, multi-line description of how this
+     * selector classifies a key: which algorithm was selected at compile time
+     * (single 8-bit window, gperf-style perfect hash, or hash-and-displace), the
+     * exact bytes/positions it inspects, and the contents of the lookup tables
+     * (which window bytes or hash slots map to which key). The text mirrors what
+     * match_raw() does step by step.
+     *
+     * Everything it reports is derived from the compile-time tables, so describe()
+     * is itself usable in a constant expression when the standard library supports
+     * constexpr std::string (__cpp_lib_constexpr_string):
+     *
+     *   static_assert(!key_selector<"name", "city">::describe().empty());
+     *
+     * It allocates a std::string and is meant for documentation, debugging and
+     * tests, not for any hot path.
+     */
+    static SIMDJSON_CONSTEXPR_STRING std::string describe() {
+        std::string s;
+        s += "key_selector: ";
+        key_selector_detail::append_uint(s, N);
+        s += " keys, max key length ";
+        key_selector_detail::append_uint(s, max_key_len);
+        s += "\nkeys:\n";
+        for (std::size_t i = 0; i < N; ++i) {
+            s += "  [";
+            key_selector_detail::append_uint(s, i);
+            s += "] \"";
+            s += keys[i];
+            s += "\" (length ";
+            key_selector_detail::append_uint(s, keys[i].size());
+            s += ")\n";
+        }
+        if constexpr (window.ok) {
+            // Mirrors the window fast path of match_raw().
+            s += "algorithm: single 8-bit window\n";
+            s += "  step 1: read 2 bytes at offset ";
+            key_selector_detail::append_uint(s, window.byte_offset);
+            s += ", interpret them as a little-endian 16-bit value, shift right by ";
+            key_selector_detail::append_uint(s, window.shift);
+            s += " bits, and keep the low 8 bits\n";
+            s += "  step 2: map that byte through a 256-entry table to a key index (";
+            key_selector_detail::append_uint(s, N);
+            s += " means no match):\n";
+            for (std::size_t b = 0; b < 256; ++b) {
+                if (window.window_to_key[b] < N) {
+                    s += "    byte ";
+                    key_selector_detail::append_byte(s, static_cast<unsigned>(b));
+                    s += " -> key ";
+                    key_selector_detail::append_uint(s, window.window_to_key[b]);
+                    s += "\n";
+                }
+            }
+            s += "  step 3: confirm the candidate by checking the closing quote sits at the key's length and comparing the key bytes\n";
+        } else {
+            // Mirrors the perfect-hash path of match_raw().
+            if constexpr (phf.num_positions == key_selector_detail::HD_MODE) {
+                s += "algorithm: hash-and-displace perfect hash\n";
+                s += "  step 1: bucket = (first_byte + last_byte*3 + length*17) mod 256\n";
+                s += "  step 2: keyhash = base-31 rolling hash of the length and the first ";
+                key_selector_detail::append_uint(s, phf.hd_hash_variant);
+                s += " bytes\n";
+                s += "  step 3: slot = (displacement[bucket] + keyhash) mod ";
+                key_selector_detail::append_uint(s, table_size);
+                s += "\n  step 4: slot_to_key[slot] gives the candidate key index\n";
+                s += "  per-key derivation:\n";
+                for (std::size_t i = 0; i < N; ++i) {
+                    std::string_view k = keys[i];
+                    std::size_t bucket = key_selector_detail::hd_bucket_hash(k);
+                    std::size_t kh = (phf.hd_hash_variant == key_selector_detail::HD_HASH_2BYTE_FLAG)
+                        ? key_selector_detail::hd_key_hash_2(k)
+                        : key_selector_detail::hd_key_hash_4(k);
+                    std::size_t slot = (phf.asso_values[0][bucket] + kh) & (table_size - 1);
+                    s += "    \"";
+                    s += k;
+                    s += "\": bucket=";
+                    key_selector_detail::append_uint(s, bucket);
+                    s += " displacement=";
+                    key_selector_detail::append_uint(s, phf.asso_values[0][bucket]);
+                    s += " keyhash=";
+                    key_selector_detail::append_uint(s, kh);
+                    s += " slot=";
+                    key_selector_detail::append_uint(s, slot);
+                    s += "\n";
+                }
+            } else {
+                s += "algorithm: gperf-style perfect hash over ";
+                key_selector_detail::append_uint(s, phf.num_positions);
+                s += " character position(s)\n";
+                s += "  step 1: h = key length\n";
+                s += "  step 2: for each position below, add its association value for the key byte there (a position past the key's length contributes 0):\n";
+                for (std::size_t i = 0; i < phf.num_positions; ++i) {
+                    s += "    position ";
+                    if (phf.positions[i] == key_selector_detail::POS_LAST_CHAR) {
+                        s += "last character";
+                    } else {
+                        s += "byte index ";
+                        key_selector_detail::append_uint(s, phf.positions[i]);
+                    }
+                    s += "\n";
+                }
+                s += "  step 3: slot = h mod ";
+                key_selector_detail::append_uint(s, table_size);
+                s += " (a power of two, applied as a bitmask)\n";
+                s += "  step 4: slot_to_key[slot] gives the candidate key index\n";
+                s += "  per-key derivation:\n";
+                for (std::size_t i = 0; i < N; ++i) {
+                    std::string_view k = keys[i];
+                    std::size_t h = k.size();
+                    for (std::size_t pi = 0; pi < phf.num_positions; ++pi) {
+                        std::size_t pos = phf.positions[pi];
+                        std::size_t idx = (pos == key_selector_detail::POS_LAST_CHAR)
+                                          ? (k.size() - 1) : pos;
+                        if (idx < k.size()) {
+                            h += phf.asso_values[pi][static_cast<unsigned char>(k[idx])];
+                        }
+                    }
+                    std::size_t slot = h & (table_size - 1);
+                    s += "    \"";
+                    s += k;
+                    s += "\": h=";
+                    key_selector_detail::append_uint(s, h);
+                    s += " slot=";
+                    key_selector_detail::append_uint(s, slot);
+                    s += "\n";
+                }
+            }
+            s += "  occupied slots (slot -> key):\n";
+            for (std::size_t slot = 0; slot < table_size; ++slot) {
+                if (phf.slot_to_key[slot] < N) {
+                    s += "    slot ";
+                    key_selector_detail::append_uint(s, slot);
+                    s += " -> key ";
+                    key_selector_detail::append_uint(s, phf.slot_to_key[slot]);
+                    s += " (\"";
+                    s += keys[phf.slot_to_key[slot]];
+                    s += "\", length ";
+                    key_selector_detail::append_uint(s, phf.slot_key_len[slot]);
+                    s += ")\n";
+                }
+            }
+            s += "  confirm the candidate by checking the key length matches and comparing the key bytes\n";
+        }
+        return s;
+    }
+};
+
+namespace key_selector_detail {
+template <typename> struct is_key_selector : std::false_type {};
+template <constevalutil::fixed_string... Keys>
+struct is_key_selector<key_selector<Keys...>> : std::true_type {};
+} // namespace key_selector_detail
+
+/**
+ * Matches any instantiation of key_selector<Keys...>. Used to constrain
+ * object::for_each so that passing a non-selector type yields a clear
+ * constraint error rather than a cascade of failures inside for_each.
+ */
+template <typename T>
+concept key_selector_type = key_selector_detail::is_key_selector<T>::value;
+
+} // namespace ondemand
+} // namespace haswell
+} // namespace simdjson
+
+#endif // SIMDJSON_SUPPORTS_CONCEPTS
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+/* end file simdjson/generic/ondemand/key_selector.h for haswell */
 /* including simdjson/generic/ondemand/object.h for haswell: #include "simdjson/generic/ondemand/object.h" */
 /* begin file simdjson/generic/ondemand/object.h for haswell */
 #ifndef SIMDJSON_GENERIC_ONDEMAND_OBJECT_H
@@ -94655,6 +119183,7 @@ public:
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/implementation_simdjson_result_base.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/key_selector.h" */
 /* amalgamation skipped (editor-only): #include <vector> */
 /* amalgamation skipped (editor-only): #if SIMDJSON_STATIC_REFLECTION && SIMDJSON_SUPPORTS_CONCEPTS */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_string_builder.h"  // for constevalutil::fixed_string */
@@ -94665,6 +119194,114 @@ namespace simdjson {
 namespace haswell {
 namespace ondemand {

+#if SIMDJSON_SUPPORTS_CONCEPTS
+/**
+ * Result of object::for_each: the first error encountered (SUCCESS if none) and
+ * the number of distinct selector keys that matched during the walk. A
+ * matched_count equal to Selector::size() means every selected key was present
+ * in the object. Implicitly converts to error_code so existing callers that only
+ * care about the error (including SIMDJSON_TRY and the test ASSERT_* macros) keep
+ * working unchanged.
+ *
+ * On error paths (a handler returns a non-SUCCESS error_code, or a direct-target
+ * deserialization via value::get fails), matched_count reflects only the number
+ * of distinct selected keys that were successfully handled *before* the failure;
+ * the failing key is not counted. This is the current behavior.
+ *
+ * Marked [[nodiscard]]: for_each surfaces parse/type errors (from a matched
+ * value, a returning handler, or the structural walk itself) only through this
+ * result, so it must not be silently dropped. This matters most in builds
+ * without exceptions, where it is the only channel for those errors. To
+ * deliberately ignore it, assign to a variable or cast to void.
+ */
+struct [[nodiscard]] for_each_result {
+  error_code error{SUCCESS};
+  std::size_t matched_count{0};
+  constexpr operator error_code() const noexcept { return error; }
+};
+
+namespace key_selector_for_each_detail {
+/**
+ * A per-key handler for the variadic object::for_each is either:
+ *   - an invocable taking a value (run custom logic for that field), or
+ *   - a deserialization target T, in which case the matched value is assigned
+ *     directly via value::get(T&).
+ * The target form lets callers bind fields straight to variables without a
+ * lambda per key -- e.g. obj.for_each<"name","city","age">(name, city, age) --
+ * while still allowing a lambda in any position when custom logic is needed
+ * (for example to descend into a nested object).
+ */
+template <typename H>
+concept field_handler =
+    std::is_invocable_v<std::remove_reference_t<H>&, value> ||
+    ::simdjson::deserializable<std::remove_cvref_t<H>, value>;
+
+/**
+ * noexcept-ness of handling a single handler: nothrow-invocability for the
+ * callback form, or nothrow-deserializability for the direct-target form
+ * (builtin scalar/string targets are noexcept; custom targets follow their
+ * own tag_invoke noexcept specification).
+ */
+template <typename H>
+inline constexpr bool nothrow_field_handler_v =
+    std::is_invocable_v<std::remove_reference_t<H>&, value>
+        ? std::is_nothrow_invocable_v<std::remove_reference_t<H>&, value>
+        : ::simdjson::nothrow_deserializable<std::remove_cvref_t<H>, value>;
+
+/**
+ * nothrow_field_handler_v folded over a whole handler pack. The variadic
+ * for_each overloads spell their conditional-noexcept specifier with this
+ * single id-expression rather than an inline fold expression. MSVC (VS18)
+ * mishandles a fold-expression that appears directly in the noexcept-specifier
+ * of an out-of-line template definition: it fails to recognize the definition's
+ * specifier as matching the in-class declaration's and reports a spurious
+ * C2382 ("redefinition; different exception specifications"). Naming the fold
+ * here keeps the specifier a plain identifier, which matches reliably.
+ */
+template <typename... Handlers>
+inline constexpr bool nothrow_field_handlers_v =
+    (nothrow_field_handler_v<Handlers> && ...);
+} // namespace key_selector_for_each_detail
+#endif
+
+/**
+ * An opaque snapshot of an object's scanning position, obtained from
+ * object::get_current_position() and consumed by object::revert_position().
+ *
+ * This bundles the raw token position with the iteration depth that was
+ * live at the moment of capture: a scalar field value (e.g. a number)
+ * unwinds back to the object's own depth once fully consumed, while a
+ * compound value (array/object) not yet fully consumed stays one level
+ * deeper. The correct depth to restore to therefore depends on what was
+ * live when the snapshot was taken, not on the object itself, which is
+ * why both pieces travel together here rather than being recomputed later.
+ *
+ * The members are private and only constructible by object itself: a
+ * hand-built value here would let revert_position() jump to an arbitrary,
+ * unvalidated position and depth, and this type carries no way to check
+ * that a given snapshot still refers to a live object. Only ever pass
+ * along a value you got from get_current_position(), and only back to the
+ * same object, before that object has been reset() or the parser has
+ * iterate()d a new document: neither invalidates a snapshot in a way this
+ * type can detect.
+ */
+struct object_position {
+  /**
+   * Default-constructed so a variable can be declared and assigned later,
+   * matching e.g. document()/object(). Not a valid position to revert to.
+   */
+  simdjson_inline object_position() noexcept = default;
+
+private:
+  token_position position{};
+  depth_t depth{};
+
+  simdjson_inline object_position(token_position position_, depth_t depth_) noexcept
+    : position(position_), depth(depth_) {}
+
+  friend class object;
+};
+
 /**
  * A forward-only JSON object field iterator.
  */
@@ -94683,8 +119320,19 @@ public:
    * Using the iterator directly is also possible but error-prone and discouraged. In particular,
    * you must dereference the iterator exactly once per iteration (before calling '++').
    * Doing otherwise is unsafe and may lead to errors. You are responsible for ensuring
+   *
+   * With SIMDJSON_DEVELOPMENT_CHECKS, the iterator locks this object while it is
+   * alive, so that reentrant access (find_field(), reset(), ...) is reported as
+   * OUT_OF_ORDER_ITERATION.
+   */
+  simdjson_inline simdjson_result<object_iterator> begin() & noexcept;
+  /**
+   * Get an iterator to the start of a temporary object, e.g., `v.get_object().begin()`.
+   *
+   * The iterator does not depend on the object instance and may outlive it, so
+   * it does not lock it.
    */
-  simdjson_inline simdjson_result<object_iterator> begin() noexcept;
+  simdjson_inline simdjson_result<object_iterator> begin() && noexcept;
   simdjson_inline simdjson_result<object_iterator> end() noexcept;
   /**
    * Look up a field by name on an object (order-sensitive). By order-sensitive, we mean that
@@ -94696,10 +119344,11 @@ public:
    *
    * ```cpp
    * simdjson::ondemand::parser parser;
-   * auto obj = parser.parse(R"( { "x": 1, "y": 2, "z": 3 } )"_padded);
-   * double z = obj.find_field("z");
-   * double y = obj.find_field("y");
-   * double x = obj.find_field("x");
+   * auto json = R"( { "x": 1, "y": 2, "z": 3 } )"_padded;
+   * auto doc = parser.iterate(json);
+   * double z = doc.find_field("z");
+   * double y = doc.find_field("y");
+   * double x = doc.find_field("x");
    * ```
    * If you have multiple fields with a matching key ({"x": 1,  "x": 1}) be mindful
    * that only one field is returned.
@@ -94772,6 +119421,100 @@ public:
   /** @overload simdjson_inline simdjson_result<value> find_field_unordered(std::string_view key) & noexcept; */
   simdjson_inline simdjson_result<value> operator[](std::string_view key) && noexcept;

+#if SIMDJSON_SUPPORTS_CONCEPTS
+  /**
+   * Walk this object once and invoke on_match(selector_index, value) for each
+   * field whose key is in the compile-time key_selector Selector, in JSON order
+   * (first occurrence of a duplicate key wins). Iteration stops once all
+   * Selector::size() keys have matched or the object ends. The value is consumed
+   * in place, so this is a low-overhead way to extract a known set of fields
+   * regardless of their order in the JSON.
+   *
+   * Like other object iteration in simdjson, for_each consumes the object by
+   * advancing the underlying iterator state; after the call the same object
+   * instance should not be used for further field access or iteration.
+   *
+   * Usage:
+   *   using sel_t = ondemand::key_selector<"id", "text", "user">;
+   *   obj.for_each<sel_t>([&](std::size_t i, ondemand::value v) {
+   *     switch (i) { case 0: ...; case 1: ...; }
+   *   });
+   *
+   * Limitations (see key_selector): each key must be at most 63 characters long,
+   * and the number of keys should be moderate (hard limit 255; a handful is
+   * best, as the compile-time perfect hash may fail or slow compilation for
+   * large key sets). The keys must be distinct, non-empty, and free of backslash, double-quote and
+   * null bytes.
+   *
+   * The callback may return either void or an error_code. When it returns an
+   * error_code, the walk stops at the first non-SUCCESS result and that error is
+   * returned, which lets the callback surface value-parse errors.
+   *
+   * This function is conditionally noexcept: it is noexcept exactly when invoking
+   * the callback is noexcept. The callback runs inside this frame, so a throwing
+   * callback (e.g. one using the exception-throwing conversions like
+   * std::string_view(value) or uint64_t(value)) makes for_each potentially
+   * throwing too -- the exception propagates to the caller instead of crossing a
+   * noexcept boundary and calling std::terminate.
+   *
+   * @returns a for_each_result holding the first error encountered while walking
+   *          the object (including any error returned by the callback, SUCCESS if
+   *          none) and the number of distinct selector keys that matched. The
+   *          result converts implicitly to error_code, so callers that only need
+   *          the error can ignore the count.
+   */
+  template <typename Selector, typename Func>
+    requires key_selector_type<Selector> &&
+             std::is_invocable_v<Func&, std::size_t, value>
+  simdjson_inline for_each_result for_each(Func&& on_match)
+      noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, value>);
+
+  /**
+   * Variadic per-key form. Provide exactly one handler per key in the Selector
+   * (compiler-enforced). Handlers are processed in JSON document order for the
+   * matching keys. Each handler is either:
+   *   - a deserialization target (a variable), in which case the matched value
+   *     is assigned to it via value::get -- no lambda required; or
+   *   - an invocable taking the ondemand::value (for custom logic such as
+   *     descending into a nested object). It may return void or error_code;
+   *     returning error_code lets you surface parse/type errors.
+   * The two styles may be mixed freely, one handler per key.
+   *
+   * Example (bind fields straight to variables):
+   *   using fields = ondemand::key_selector<"name", "city", "age">;
+   *   obj.for_each<fields>(name, city, age);
+   *
+   * Example (mixing a target and a lambda):
+   *   obj.for_each<ondemand::key_selector<"id", "user">>(
+   *     id,                                          // assigned via value::get
+   *     [&](ondemand::value v){ u = read_user(v); }  // custom logic
+   *   );
+   *
+   * The index-based single-callback form (taking (size_t, value)) remains
+   * available for shared-state or more complex per-key logic.
+   */
+  template <typename Selector, typename... Handlers>
+    requires key_selector_type<Selector> &&
+             (sizeof...(Handlers) == Selector::size()) &&
+             (key_selector_for_each_detail::field_handler<Handlers> && ...)
+  simdjson_inline for_each_result for_each(Handlers&&... on_match)
+      noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
+  /**
+   * Direct-key shorthand. Equivalent to for_each<key_selector<Keys...>>(...).
+   * Lets you write the keys inline without a separate using/alias, binding each
+   * field straight to a variable (or a lambda, see the Selector form above):
+   *
+   *   obj.for_each<"name", "city", "age">(name, city, age);
+   */
+  template <constevalutil::fixed_string... Keys, typename... Handlers>
+    requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+             (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+             (key_selector_for_each_detail::field_handler<Handlers> && ...)
+  simdjson_inline for_each_result for_each(Handlers&&... on_match)
+      noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+#endif
+
   /**
    * Get the value associated with the given JSON pointer. We use the RFC 6901
    * https://tools.ietf.org/html/rfc6901 standard, interpreting the current node
@@ -94848,6 +119591,34 @@ public:
    * @returns true if the object contains some elements (not empty)
    */
   inline simdjson_result<bool> reset() & noexcept;
+  /**
+   * Get an opaque token representing the object's current scanning position.
+   * Pass it to revert_position() to return to this exact point later, without
+   * paying the cost of a full reset() and re-scan from the beginning.
+   *
+   * A typical use is an optional field that may or may not be next: capture
+   * the position, attempt find_field(), and on NO_SUCH_FIELD, revert_position()
+   * instead of reset() so that fields already consumed are not rescanned.
+   *
+   * The returned token is only valid for this object, and only until it is
+   * reset() or the parser iterate()s a new document; using it after either
+   * is undefined behavior (see object_position).
+   *
+   * @returns An opaque position token.
+   */
+  simdjson_inline object_position get_current_position() const noexcept;
+  /**
+   * Return the object's scanning position to a snapshot previously obtained
+   * from get_current_position(). Unlike reset(), this does not rescan the
+   * object from the beginning: fields before the captured position remain
+   * consumed, and scanning resumes exactly where the snapshot was captured.
+   *
+   * @param position A snapshot previously returned by get_current_position(),
+   *        for this same object.
+   * @returns SUCCESS, or OUT_OF_ORDER_ITERATION if called during active
+   *          iteration (SIMDJSON_DEVELOPMENT_CHECKS builds only).
+   */
+  simdjson_inline error_code revert_position(object_position position) noexcept;
   /**
    * This method scans the beginning of the object and checks whether the
    * object is empty.
@@ -94893,7 +119664,7 @@ public:
    */
   template <typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out)
-     noexcept(custom_deserializable<T, object> ? nothrow_custom_deserializable<T, object> : true) {
+     noexcept(nothrow_gettable<T, object>) {
     static_assert(custom_deserializable<T, object>);
     return deserialize(*this, out);
   }
@@ -94905,7 +119676,7 @@ public:
    */
   template <typename T>
   simdjson_inline simdjson_result<T> get()
-    noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+    noexcept(nothrow_gettable<T, object>)
   {
     static_assert(std::is_default_constructible<T>::value, "The specified type is not default constructible.");
     T out{};
@@ -94957,10 +119728,18 @@ protected:
   simdjson_warn_unused simdjson_inline error_code find_field_raw(const std::string_view key) noexcept;

   value_iterator iter{};
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  bool locked{false};
+  simdjson_inline void set_locked(bool _locked) noexcept;
+#endif

   friend class value;
   friend class document;
   friend struct simdjson_result<object>;
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  friend class object_iterator;
+  friend struct simdjson_result<object_iterator>;
+#endif
 };

 } // namespace ondemand
@@ -94976,7 +119755,8 @@ public:
   simdjson_inline simdjson_result(error_code error) noexcept; ///< @private
   simdjson_inline simdjson_result() noexcept = default;

-  simdjson_inline simdjson_result<haswell::ondemand::object_iterator> begin() noexcept;
+  simdjson_inline simdjson_result<haswell::ondemand::object_iterator> begin() & noexcept;
+  simdjson_inline simdjson_result<haswell::ondemand::object_iterator> begin() && noexcept;
   simdjson_inline simdjson_result<haswell::ondemand::object_iterator> end() noexcept;
   simdjson_inline simdjson_result<haswell::ondemand::value> find_field(std::string_view key) & noexcept;
   simdjson_inline simdjson_result<haswell::ondemand::value> find_field(std::string_view key) && noexcept;
@@ -94994,6 +119774,8 @@ public:
 #endif
   simdjson_inline error_code for_each_at_path_with_wildcard(std::string_view json_path, Func&& callback) noexcept;
   inline simdjson_result<bool> reset() noexcept;
+  inline simdjson_result<haswell::ondemand::object_position> get_current_position() noexcept;
+  inline error_code revert_position(haswell::ondemand::object_position position) noexcept;
   inline simdjson_result<bool> is_empty() noexcept;
   inline simdjson_result<size_t> count_fields() & noexcept;
   inline simdjson_result<std::string_view> raw_json() noexcept;
@@ -95001,7 +119783,7 @@ public:
   // TODO: move this code into object-inl.h

   template<typename T>
-  simdjson_inline simdjson_result<T> get() noexcept {
+  simdjson_inline simdjson_result<T> get() noexcept(nothrow_gettable<T, haswell::ondemand::object>) {
     if (error()) { return error(); }
     if constexpr (std::is_same_v<T, haswell::ondemand::object>) {
       return first;
@@ -95009,7 +119791,7 @@ public:
     return first.get<T>();
   }
   template<typename T>
-  simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept {
+  simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept(nothrow_gettable<T, haswell::ondemand::object>) {
     if (error()) { return error(); }
     if constexpr (std::is_same_v<T, haswell::ondemand::object>) {
       out = first;
@@ -95019,6 +119801,39 @@ public:
     return SUCCESS;
   }

+  /**
+   * Forwards to object::for_each on the underlying object, so error-code-style
+   * chains (e.g. doc["x"].get_object()) can call for_each without first
+   * extracting the object. If this result holds an error, that error is returned
+   * (with a zero match count) and the callback is not invoked. See
+   * object::for_each for the semantics.
+   */
+  template <typename Selector, typename Func>
+    requires haswell::ondemand::key_selector_type<Selector> &&
+             std::is_invocable_v<Func&, std::size_t, haswell::ondemand::value>
+  simdjson_inline haswell::ondemand::for_each_result for_each(Func&& on_match)
+      noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, haswell::ondemand::value>);
+
+  /**
+   * Forwarding overload for the variadic per-key form (targets and/or lambdas).
+   */
+  template <typename Selector, typename... Handlers>
+    requires haswell::ondemand::key_selector_type<Selector> &&
+             (sizeof...(Handlers) == Selector::size()) &&
+             (haswell::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+  simdjson_inline haswell::ondemand::for_each_result for_each(Handlers&&... on_match)
+      noexcept(haswell::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
+  /**
+   * Forwarding overload for the direct-key variadic form.
+   */
+  template <constevalutil::fixed_string... Keys, typename... Handlers>
+    requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+             (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+             (haswell::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+  simdjson_inline haswell::ondemand::for_each_result for_each(Handlers&&... on_match)
+      noexcept(haswell::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
 #if SIMDJSON_STATIC_REFLECTION
   // TODO: move this code into object-inl.h
   template<constevalutil::fixed_string... FieldNames, typename T>
@@ -95059,6 +119874,15 @@ public:
    */
   simdjson_inline object_iterator() noexcept = default;

+#if SIMDJSON_DEVELOPMENT_CHECKS
+   simdjson_inline ~object_iterator() noexcept;
+
+   simdjson_inline object_iterator(object_iterator&&) noexcept;
+   simdjson_inline object_iterator& operator=(object_iterator&&) noexcept;
+   simdjson_inline object_iterator(const object_iterator&) noexcept;
+   simdjson_inline object_iterator& operator=(const object_iterator&) noexcept;
+#endif
+
   //
   // Iterator interface
   //
@@ -95078,6 +119902,9 @@ public:
 private:
 #if SIMDJSON_DEVELOPMENT_CHECKS
    bool has_been_referenced{false};
+   object* parent{nullptr};
+
+   simdjson_inline object_iterator(const value_iterator &_iter, object* _parent) noexcept;
 #endif
   /**
    * The underlying JSON iterator.
@@ -95123,6 +119950,191 @@ public:

 #endif // SIMDJSON_GENERIC_ONDEMAND_OBJECT_ITERATOR_H
 /* end file simdjson/generic/ondemand/object_iterator.h for haswell */
+/* including simdjson/generic/ondemand/ranges.h for haswell: #include "simdjson/generic/ondemand/ranges.h" */
+/* begin file simdjson/generic/ondemand/ranges.h for haswell */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_RANGES_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_ONDEMAND_RANGES_H */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/field.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#if SIMDJSON_SUPPORTS_RANGES
+
+namespace simdjson {
+namespace haswell {
+namespace ondemand {
+
+/**
+ * A ranges-compatible iterator adapter for JSON arrays.
+ *
+ * Wraps array_iterator to satisfy std::input_iterator by providing:
+ * - const operator* (via mutable internal state)
+ * - post-increment operator
+ * - iterator_concept tag
+ *
+ * The mutable approach is standard for single-pass input iterators that
+ * read from external sources (similar to std::istream_iterator).
+ */
+class array_range_iterator {
+public:
+  using iterator_concept = std::input_iterator_tag;
+  using value_type = simdjson_result<value>;
+  using reference = simdjson_result<value>;
+  using difference_type = std::ptrdiff_t;
+
+  simdjson_inline array_range_iterator() noexcept = default;
+  simdjson_inline explicit array_range_iterator(array_iterator iter) noexcept;
+
+  /**
+   * Get the current element. Const-qualified for std::indirectly_readable;
+   * internally delegates to the mutable wrapped iterator.
+   */
+  simdjson_inline simdjson_result<value> operator*() const noexcept;
+  simdjson_inline array_range_iterator& operator++() noexcept;
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+  simdjson_inline void operator++(int) noexcept;
+SIMDJSON_POP_DISABLE_WARNINGS
+  /**
+   * Comparison delegates to array_iterator::operator==, which checks
+   * whether the underlying parser has finished the array (depth-based).
+   */
+  simdjson_inline friend bool operator==(const array_range_iterator& a,
+                                         const array_range_iterator& b) noexcept {
+    return a.iter_ == b.iter_;
+  }
+
+private:
+  mutable array_iterator iter_{};
+};
+
+/**
+ * A std::ranges::view over a JSON array.
+ *
+ * Wraps an ondemand::array and exposes begin()/end() that return
+ * array_range_iterator (satisfying std::input_iterator), enabling
+ * use with std::views::transform and other range adaptors.
+ *
+ * If the array's begin() returns an error (only possible under
+ * SIMDJSON_DEVELOPMENT_CHECKS), the range will be empty and error()
+ * will return the error code.
+ *
+ * Usage:
+ *   ondemand::parser parser;
+ *   auto doc = parser.iterate(json);
+ *   auto arr = doc.get_array().value();
+ *   for (auto elem : ondemand::get_range(arr)) { ... }
+ */
+class array_range {
+public:
+  simdjson_inline array_range() noexcept = default;
+  simdjson_inline explicit array_range(array& arr) noexcept;
+
+  simdjson_inline array_range_iterator begin() noexcept;
+  simdjson_inline array_range_iterator end() noexcept;
+
+  /** Returns SUCCESS if the range was created successfully, or the error code otherwise. */
+  simdjson_inline error_code error() const noexcept { return error_; }
+
+private:
+  array_iterator begin_{};
+  array_iterator end_{};
+  error_code error_{SUCCESS};
+};
+
+/**
+ * A ranges-compatible iterator adapter for JSON objects.
+ *
+ * Wraps object_iterator to satisfy std::input_iterator, yielding
+ * simdjson_result<field> elements (key-value pairs).
+ */
+class object_range_iterator {
+public:
+  using iterator_concept = std::input_iterator_tag;
+  using value_type = simdjson_result<field>;
+  using reference = simdjson_result<field>;
+  using difference_type = std::ptrdiff_t;
+
+  simdjson_inline object_range_iterator() noexcept = default;
+  simdjson_inline explicit object_range_iterator(object_iterator iter) noexcept;
+
+  simdjson_inline simdjson_result<field> operator*() const noexcept;
+  simdjson_inline object_range_iterator& operator++() noexcept;
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+  simdjson_inline void operator++(int) noexcept;
+SIMDJSON_POP_DISABLE_WARNINGS
+  simdjson_inline friend bool operator==(const object_range_iterator& a,
+                                         const object_range_iterator& b) noexcept {
+    return a.iter_ == b.iter_;
+  }
+
+private:
+  mutable object_iterator iter_{};
+};
+
+/**
+ * A std::ranges::view over a JSON object.
+ *
+ * Wraps an ondemand::object and exposes begin()/end() that return
+ * object_range_iterator, enabling use with range adaptors.
+ *
+ * If the object's begin() returns an error, the range will be empty
+ * and error() will return the error code.
+ */
+class object_range {
+public:
+  simdjson_inline object_range() noexcept = default;
+  simdjson_inline explicit object_range(object& obj) noexcept;
+
+  simdjson_inline object_range_iterator begin() noexcept;
+  simdjson_inline object_range_iterator end() noexcept;
+
+  /** Returns SUCCESS if the range was created successfully, or the error code otherwise. */
+  simdjson_inline error_code error() const noexcept { return error_; }
+
+private:
+  object_iterator begin_{};
+  object_iterator end_{};
+  error_code error_{SUCCESS};
+};
+
+/** Get a std::ranges compatible view over a JSON array. */
+simdjson_inline array_range get_range(array& arr) noexcept;
+
+/** Get a std::ranges compatible view over a JSON object (key-value pairs). */
+simdjson_inline object_range get_key_value_range(object& obj) noexcept;
+
+#if SIMDJSON_EXCEPTIONS
+/** Get a std::ranges compatible view, unwrapping the simdjson_result (throws on error). */
+simdjson_inline array_range get_range(simdjson_result<array> result);
+
+/** Get a std::ranges compatible view, unwrapping the simdjson_result (throws on error). */
+simdjson_inline object_range get_key_value_range(simdjson_result<object> result);
+#endif // SIMDJSON_EXCEPTIONS
+
+} // namespace ondemand
+} // namespace haswell
+} // namespace simdjson
+
+namespace std {
+namespace ranges {
+template<>
+inline constexpr bool enable_view<simdjson::haswell::ondemand::array_range> = true;
+template<>
+inline constexpr bool enable_view<simdjson::haswell::ondemand::object_range> = true;
+} // namespace ranges
+} // namespace std
+
+#endif // SIMDJSON_SUPPORTS_RANGES
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_RANGES_H
+/* end file simdjson/generic/ondemand/ranges.h for haswell */
 /* including simdjson/generic/ondemand/serialization.h for haswell: #include "simdjson/generic/ondemand/serialization.h" */
 /* begin file simdjson/generic/ondemand/serialization.h for haswell */
 #ifndef SIMDJSON_GENERIC_ONDEMAND_SERIALIZATION_H
@@ -95255,12 +120267,14 @@ inline std::ostream& operator<<(std::ostream& out, simdjson::simdjson_result<sim
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/annotations.h" */
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */

 #include <concepts>
 #include <limits>
 #if SIMDJSON_STATIC_REFLECTION
 #include <meta>
+#include <vector>
 // #include <static_reflection> // for std::define_static_string - header not available yet
 #endif

@@ -95285,10 +120299,25 @@ error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept {

 template <std::floating_point T>
 error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept {
-  double x;
-  SIMDJSON_TRY(val.get_double().get(x));
-  out = static_cast<T>(x);
-  return SUCCESS;
+  if constexpr (std::is_same_v<T, float>) {
+    // Going through binary64 and then rounding to binary32 would round twice
+    // and could produce a value that is not the float nearest to the JSON
+    // number, so we parse to binary32 directly.
+    return val.get_float().get(out);
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  } else if constexpr (std::is_same_v<T, std::float32_t>) {
+    // Same reason as float.
+    float x;
+    SIMDJSON_TRY(val.get_float().get(x));
+    out = static_cast<T>(x);
+    return SUCCESS;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+  } else {
+    double x;
+    SIMDJSON_TRY(val.get_double().get(x));
+    out = static_cast<T>(x);
+    return SUCCESS;
+  }
 }

 template <std::signed_integral T>
@@ -95324,11 +120353,59 @@ template <concepts::constructible_from_string_view T, typename ValT>
 error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(std::is_nothrow_constructible_v<T, std::string_view>) {
   std::string_view str;
   SIMDJSON_TRY(val.get_string().get(str));
-  out = T{str};
+  if constexpr (requires { out.assign(str.data(), str.size()); }) {
+    // Copy straight into out (e.g., std::string): building a temporary and
+    // move-assigning it is markedly slower.
+    out.assign(str.data(), str.size());
+  } else {
+    out = T{str};
+  }
+  return SUCCESS;
+}
+
+#if SIMDJSON_SUPPORTS_CHAR8_T
+// any C++20 char8_t string-like type (can be constructed from std::u8string_view),
+// such as std::u8string
+template <concepts::constructible_from_u8string_view T, typename ValT>
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(std::is_nothrow_constructible_v<T, std::u8string_view>) {
+  std::u8string_view str;
+  SIMDJSON_TRY(val.get_u8string().get(str));
+  if constexpr (requires { out.assign(str.data(), str.size()); }) {
+    // Copy straight into out (e.g., std::u8string), as for std::string above.
+    out.assign(str.data(), str.size());
+  } else {
+    out = T{str};
+  }
   return SUCCESS;
 }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T


+namespace details {
+// Whether to deserialize the elements of the container T directly into a new
+// element (emplace_one(out) and then get) rather than into a temporary that is
+// then moved into the container. A temporary of a trivially copyable type lives
+// in registers and is cheap to move, and value-initializing a small element in
+// place costs more than it saves (GCC zeroes it with rep stos). But moving a
+// large element, or a string (whose deserialization then needs a temporary
+// string and a move assignment of its own), is expensive.
+template <typename T>
+concept deserialize_in_place =
+    concepts::returns_reference<T> && requires(T &c) { c.pop_back(); } &&
+    !std::is_trivially_copyable_v<typename T::value_type> &&
+    (sizeof(typename T::value_type) > 32 || concepts::constructible_from_string_view<typename T::value_type>);
+
+// Removes the last element of the container on destruction while armed.
+template <typename T>
+struct pop_back_guard {
+  T &container;
+  bool armed{true};
+  ~pop_back_guard() {
+    if (armed) { container.pop_back(); }
+  }
+};
+} // namespace details
+
 /**
  * STL containers have several constructors including one that takes a single
  * size argument. Thus, some compilers (Visual Studio) will not be able to
@@ -95352,22 +120429,65 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
     SIMDJSON_TRY(val.get_array().get(arr));
   }

-  for (auto v : arr) {
-    if constexpr (concepts::returns_reference<T>) {
-      if (auto const err = v.get<value_type>().get(concepts::emplace_one(out));
-          err) {
-        // If an error occurs, the empty element that we just inserted gets
-        // removed. We're not using a temp variable because if T is a heavy
-        // type, we want the valid path to be the fast path and the slow path be
-        // the path that has errors in it.
-        if constexpr (requires { out.pop_back(); }) {
-          static_cast<void>(out.pop_back());
+  if constexpr (std::is_same_v<T, std::vector<value_type>> && !std::is_same_v<value_type, bool>) {
+    // Collect the elements in a per-thread scratch vector that keeps its
+    // capacity from call to call, then move them into out after reserving the
+    // exact size: out is allocated once instead of being regrown. A nested
+    // array of the same type finds the scratch busy and takes the paths below.
+    // Prior related work: jsonifier keeps a thread-local vector and sizes the
+    // caller's vector from that element count (parse_impl.hpp,
+    // https://github.com/nihilai-collective/Jsonifier).
+    struct scratch_space {
+      std::vector<value_type> elements{};
+      bool busy{false};
+    };
+    static thread_local scratch_space scratch;
+    if (!scratch.busy && out.empty()) {
+      struct release_scratch {
+        scratch_space &s;
+        T &out;
+        size_t parsed{0};
+        bool complete{false};
+        // On an error or an exception, out gets the elements parsed so far (as
+        // with the loops below), without allocating. Kept out of the hot path.
+        simdjson_never_inline void keep_parsed() noexcept {
+          s.elements.resize(parsed);
+          out.swap(s.elements);
         }
-        return err;
-      }
-    } else {
+        ~release_scratch() {
+          if (simdjson_unlikely(!complete)) { keep_parsed(); }
+          s.elements.clear();
+          // Do not hold on to the memory of a very large array.
+          if (s.elements.capacity() * sizeof(value_type) > (1 << 20)) { std::vector<value_type>().swap(s.elements); }
+          s.busy = false;
+        }
+      } release{scratch, out};
+      scratch.busy = true;
+      for (auto v : arr) {
+        SIMDJSON_TRY(v.get<value_type>(scratch.elements.emplace_back()));
+        release.parsed++;
+      }
+      out.reserve(release.parsed);
+      release.complete = true;
+      for (auto &e : scratch.elements) { out.emplace_back(std::move(e)); }
+      return SUCCESS;
+    }
+  }
+  if constexpr (details::deserialize_in_place<T>) {
+    for (auto v : arr) {
+      auto &slot = concepts::emplace_one(out);
+      // An error or an exception (a user tag_invoke may throw) must not leave
+      // a partially deserialized element behind.
+      details::pop_back_guard<T> guard{out};
+      SIMDJSON_TRY(v.get<value_type>(slot));
+      guard.armed = false;
+    }
+  } else {
+    for (auto v : arr) {
+      // Deserialize into a temporary first: an error or an exception (a user
+      // tag_invoke may throw) must not leave a default-constructed element behind.
       value_type temp;
-      if (auto const err = v.get<value_type>().get(temp); err) {
+      if (auto const err = v.get<value_type>(temp); err) {
         return err;
       }
       concepts::emplace_one(out, std::move(temp));
@@ -95408,7 +120528,7 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
 }

 template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, haswell::ondemand::object &obj, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, haswell::ondemand::object &obj, T &out) noexcept(false) {
   using value_type = typename std::remove_cvref_t<T>::mapped_type;

   out.clear();
@@ -95427,21 +120547,21 @@ error_code tag_invoke(deserialize_tag, haswell::ondemand::object &obj, T &out) n
 }

 template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, haswell::ondemand::value &val, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, haswell::ondemand::value &val, T &out) noexcept(false) {
   haswell::ondemand::object obj;
   SIMDJSON_TRY(val.get_object().get(obj));
   return simdjson::deserialize(obj, out);
 }

 template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, haswell::ondemand::document &doc, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, haswell::ondemand::document &doc, T &out) noexcept(false) {
   haswell::ondemand::object obj;
   SIMDJSON_TRY(doc.get_object().get(obj));
   return simdjson::deserialize(obj, out);
 }

 template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, haswell::ondemand::document_reference &doc, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, haswell::ondemand::document_reference &doc, T &out) noexcept(false) {
   haswell::ondemand::object obj;
   SIMDJSON_TRY(doc.get_object().get(obj));
   return simdjson::deserialize(obj, out);
@@ -95452,10 +120572,6 @@ error_code tag_invoke(deserialize_tag, haswell::ondemand::document_reference &do
  * This CPO (Customization Point Object) will help deserialize into
  * smart pointers.
  *
- * If constructing T is nothrow, this conversion should be nothrow as well since
- * we return MEMALLOC if we're not able to allocate memory instead of throwing
- * the error message.
- *
  * @tparam T The type inside the smart pointer
  * @tparam ValT document/value type
  * @param val document/value
@@ -95463,7 +120579,7 @@ error_code tag_invoke(deserialize_tag, haswell::ondemand::document_reference &do
  * @return status of the conversion
  */
 template <concepts::smart_pointer T, typename ValT>
-error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(nothrow_deserializable<typename std::remove_cvref_t<T>::element_type, ValT>) {
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
   using element_type = typename std::remove_cvref_t<T>::element_type;

   // For better error messages, don't use these as constraints on
@@ -95475,12 +120591,13 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(nothrow_deser
       std::is_default_constructible_v<element_type>,
       "The specified type inside the unique_ptr must default constructible.");

-  auto ptr = new (std::nothrow) element_type();
-  if (ptr == nullptr) {
+  // Own the allocation before get(): a user tag_invoke may throw.
+  std::unique_ptr<element_type> ptr(new (std::nothrow) element_type());
+  if (!ptr) {
     return MEMALLOC;
   }
   SIMDJSON_TRY(val.template get<element_type>(*ptr));
-  out.reset(ptr);
+  out = std::move(ptr);
   return SUCCESS;
 }

@@ -95512,53 +120629,595 @@ error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept(nothrow_deser

 template <typename T>
 constexpr bool user_defined_type = (std::is_class_v<T>
-&& !std::is_same_v<T, std::string> && !std::is_same_v<T, std::string_view> && !concepts::optional_type<T> &&
-!concepts::appendable_containers<T>);
+&& !std::is_same_v<T, std::string> && !std::is_same_v<T, std::string_view>
+#if SIMDJSON_SUPPORTS_CHAR8_T
+// The char8_t string types are class types with no reflectable members, so
+// without this they would be taken for user structs and the reflection
+// overload below would compete with the u8 string overload in
+// std_deserialize.h, making every tag_invoke call on them ambiguous.
+&& !std::is_same_v<T, std::u8string> && !std::is_same_v<T, std::u8string_view>
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+&& !concepts::optional_type<T> &&
+!concepts::appendable_containers<T>
+// simdjson's own types (array, object, value, raw_json_string, number,
+// document, document_reference) have dedicated get<T>() specializations and
+// must never go through reflection.
+&& !is_builtin_deserializable_v<T>
+&& !std::is_same_v<T, haswell::ondemand::number>
+&& !std::is_same_v<T, haswell::ondemand::document>
+&& !std::is_same_v<T, haswell::ondemand::document_reference>);
+
+
+// key_selector_reflection_detail is defined unconditionally (it only requires
+// static reflection). It provides the compile-time machinery for building a
+// key_selector from a struct's members, the per-member helpers shared by every
+// deserialization path (they implement the annotations of annotations.h), the
+// ordered per-member path used by the opt-out build (see
+// deserialize_struct_ordered below), and the scan used for structs annotated
+// with deny_unknown_fields and as an automatic fallback (see
+// deserialize_struct_scan and keys_fit_selector below).
+namespace key_selector_reflection_detail {
+
+// A member participates if it is public, non-const, and not annotated with skip
+// or skip_deserializing.
+consteval bool is_eligible_member(std::meta::info mem) {
+  return !std::meta::is_const(mem) && std::meta::is_public(mem)
+      && !simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_tag)
+      && !simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_deserializing_tag);
+}
+
+// A field is a data member reached from T through a chain of members: usually a
+// direct member of T (a path of length one), or a member of a member annotated
+// with flatten (the members of a flattened structure are fields of the enclosing
+// one). A field is identified by the type member_path<m1, m2, ..., leaf>.
+template <std::meta::info First, std::meta::info... Rest>
+struct member_path {
+  // The data member holding the value; its annotations drive (de)serialization.
+  static constexpr std::meta::info leaf = [] {
+    std::meta::info members[] = {First, Rest...};
+    return members[sizeof...(Rest)];
+  }();
+  template <typename T>
+  static simdjson_inline constexpr auto &get(T &obj) noexcept {
+    if constexpr (sizeof...(Rest) == 0) {
+      return obj.[:First:];
+    } else {
+      return member_path<Rest...>::get(obj.[:First:]);
+    }
+  }
+};
+
+// True when T can be flattened: a structure deserialized member by member (not
+// a string, a container, an optional, a smart pointer, ...).
+template <typename T>
+constexpr bool flattenable_type = user_defined_type<T> && !concepts::string_view_keyed_map<T>
+    && !concepts::container_but_not_string<T> && !concepts::smart_pointer<T>;
+
+consteval void append_eligible_fields(std::meta::info type, std::vector<std::meta::info> &prefix,
+                                      std::vector<std::meta::info> &fields) {
+  for (std::meta::info mem : std::meta::nonstatic_data_members_of(type, std::meta::access_context::unchecked())) {
+    if (!is_eligible_member(mem)) { continue; }
+    prefix.push_back(std::meta::reflect_constant(mem));
+    if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::flatten_tag)) {
+      std::meta::info flattened = simdjson::detail::flattened_type(mem);
+      if (!std::meta::extract<bool>(std::meta::substitute(^^flattenable_type, {flattened}))) {
+        throw std::meta::exception(u8"simdjson::flatten requires a member whose type is a structure deserialized member by member", mem);
+      }
+      append_eligible_fields(flattened, prefix, fields);
+    } else {
+      fields.push_back(std::meta::substitute(^^member_path, prefix));
+    }
+    prefix.pop_back();
+  }
+}
+
+// The fields of `type` that participate in deserialization, in declaration order
+// (the fields of a flattened member take its place). A field's position in this
+// list is its "field index" below.
+consteval std::vector<std::meta::info> eligible_fields(std::meta::info type) {
+  std::vector<std::meta::info> prefix;
+  std::vector<std::meta::info> fields;
+  append_eligible_fields(type, prefix, fields);
+  return fields;
+}
+
+// The data member at the end of a field path.
+consteval std::meta::info field_leaf(std::meta::info path) {
+  return std::meta::extract<std::meta::info>(std::meta::template_arguments_of(path).back());
+}
+
+// Number of fields that participate in deserialization. A class can have zero
+// eligible fields (e.g. std::chrono::time_point, whose only data member is
+// private): an empty key_selector cannot be built, so the tag_invoke below
+// special-cases this count.
+template <typename T>
+consteval std::size_t eligible_field_count() {
+  return eligible_fields(^^T).size();
+}
+
+// The keys accepted for a member (or an enumerator): its JSON key followed by its
+// aliases. They are static strings, so they can be used as template arguments.
+consteval std::vector<const char *> accepted_keys_of(std::meta::info entity) {
+  std::vector<const char *> keys;
+  for (std::string_view key : simdjson::detail::json_key_names(entity)) {
+    bool repeated = false;
+    for (const char *previous : keys) { repeated = repeated || std::string_view(previous) == key; }
+    if (!repeated) { keys.push_back(std::define_static_string(key)); }
+  }
+  return keys;
+}
+
+// Every key accepted for T: the eligible fields in order, each followed by its
+// aliases. This is the key list of T's key_selector.
+consteval std::vector<const char *> accepted_keys(std::meta::info type) {
+  std::vector<const char *> keys;
+  for (std::meta::info path : eligible_fields(type)) {
+    for (const char *key : accepted_keys_of(field_leaf(path))) { keys.push_back(key); }
+  }
+  return keys;
+}
+
+// The field index of each entry of accepted_keys(type).
+consteval std::vector<std::size_t> accepted_key_fields(std::meta::info type) {
+  std::vector<std::size_t> key_fields;
+  std::vector<std::meta::info> fields = eligible_fields(type);
+  for (std::size_t i = 0; i < fields.size(); ++i) {
+    for (std::size_t j = 0; j < accepted_keys_of(field_leaf(fields[i])).size(); ++j) { key_fields.push_back(i); }
+  }
+  return key_fields;
+}
+
+// True when no two eligible fields of T accept the same key. Otherwise one JSON
+// key would have to fill several members: this is reported at compile time.
+template <typename T>
+consteval bool accepted_keys_are_distinct() {
+  std::vector<const char *> keys = accepted_keys(^^T);
+  for (std::size_t i = 0; i < keys.size(); ++i) {
+    for (std::size_t j = i + 1; j < keys.size(); ++j) {
+      if (std::string_view(keys[i]) == std::string_view(keys[j])) { return false; }
+    }
+  }
+  return true;
+}
+
+// True when some accepted key of T is written with escape sequences in JSON (a
+// double quote, a backslash or a control character). Such a key can only be
+// matched by comparing unescaped keys (deserialize_struct_scan): obj[key] and
+// the key_selector compare the raw bytes.
+template <typename T>
+consteval bool keys_need_unescaping() {
+  for (std::string_view key : accepted_keys(^^T)) {
+    for (char c : key) {
+      if (c == '\\' || c == '"' || static_cast<unsigned char>(c) < 0x20) { return true; }
+    }
+  }
+  return false;
+}

+// True when some eligible field of T has aliases: several selector keys may then
+// map to the same field.
+template <typename T>
+consteval bool has_aliases() {
+  return accepted_keys(^^T).size() != eligible_fields(^^T).size();
+}
+
+// True when `mem` is annotated with default_value or default_from (directly, or
+// through a default_value annotation on the enclosing structure).
+template <auto mem>
+consteval bool has_default() {
+  return simdjson::detail::has_annotation(mem, ^^simdjson::detail::default_value_tag)
+      || simdjson::detail::has_annotation(std::meta::parent_of(mem), ^^simdjson::detail::default_value_tag)
+      || simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::default_from_t) != std::meta::info{};
+}
+
+// True when a missing key for `mem` is not an error: optional members and
+// members with a default.
+template <auto mem>
+consteval bool may_be_absent() {
+  return concepts::optional_type<typename [: std::meta::type_of(mem) :]> || has_default<mem>();
+}
+
+// True when every eligible field of T is required (none may be absent). In that
+// case presence can be checked with a single match count instead of a per-field
+// "seen" array.
+template <typename T>
+consteval bool all_eligible_fields_required() {
+  bool all_required = true;
+  template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+    if constexpr (may_be_absent<[: path :]::leaf>()) {
+      all_required = false;
+    }
+  }
+  return all_required;
+}
+
+// `key` as a constevalutil::fixed_string usable as an NTTP.
+template <const char *key>
+consteval auto key_fixed_string() {
+  constexpr std::string_view key_view{ key };
+  char buffer[key_view.size() + 1] = {};
+  for (std::size_t i = 0; i < key_view.size(); ++i) { buffer[i] = key_view[i]; }
+  return constevalutil::fixed_string<key_view.size() + 1>(buffer);
+}
+
+// key_selector template arguments (one fixed_string per accepted key), in the
+// order of accepted_keys.
+template <typename T>
+consteval std::vector<std::meta::info> selector_key_args() {
+  std::vector<std::meta::info> args;
+  template for (constexpr const char *key : std::define_static_array(accepted_keys(^^T))) {
+    args.push_back(std::meta::reflect_constant(key_fixed_string<key>()));
+  }
+  return args;
+}
+
+// key_selector whose keys are exactly T's accepted keys (index i <-> i-th key).
+template <typename T>
+using selector_for = typename [: std::meta::substitute(
+    ^^haswell::ondemand::key_selector, selector_key_args<T>()) :];
+
+// True when T's accepted keys satisfy every key_selector requirement, so a
+// key_selector can be built for T without a compile-time error. This mirrors the
+// key_selector limits (see key_selector.h): at most 255 keys, each key non-empty
+// and at most 63 characters, no backslash / double-quote / null byte, and all
+// keys distinct. Keys with any other control character are excluded too: they
+// are escaped in JSON, so their raw bytes never match. When this returns false
+// the deserializer falls back to deserialize_struct_scan instead of failing to
+// compile.
+template <typename T>
+consteval bool keys_fit_selector() {
+  std::vector<std::string_view> keys;
+  for (const char *key : accepted_keys(^^T)) { keys.push_back(key); }
+  if (keys.size() > 255) { return false; }
+  for (std::size_t i = 0; i < keys.size(); ++i) {
+    if (keys[i].empty() || keys[i].size() > 63) { return false; }
+    for (char c : keys[i]) {
+      if (c == '\\' || c == '"' || static_cast<unsigned char>(c) < 0x20) { return false; }
+    }
+    for (std::size_t j = i + 1; j < keys.size(); ++j) {
+      if (keys[i] == keys[j]) { return false; }
+    }
+  }
+  return true;
+}
+
+// True when the class `adapter` declares a member named deserialize.
+consteval bool declares_deserialize(std::meta::info adapter) {
+  for (std::meta::info m : std::meta::members_of(adapter, std::meta::access_context::unchecked())) {
+    if (std::meta::has_identifier(m) && std::meta::identifier_of(m) == "deserialize") { return true; }
+  }
+  return false;
+}
+
+// Deserialize a JSON value into `target`, the storage of member `mem`, through
+// the member's with<Adapter> annotation when the adapter provides a deserialize
+// function.
+template <auto mem, typename ValueT, typename M>
+simdjson_warn_unused simdjson_inline error_code deserialize_member_value(ValueT &field_value, M &target) noexcept(false) {
+  constexpr std::meta::info with_type = simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::with_t);
+  if constexpr (with_type != std::meta::info{}) {
+    using adapter = typename [: with_type :]::adapter;
+    using ondemand_value = haswell::ondemand::value;
+    if constexpr (requires { { adapter::deserialize(field_value, target) } -> std::convertible_to<error_code>; }) {
+      return adapter::deserialize(field_value, target);
+    } else if constexpr (requires(ondemand_value &v) { { adapter::deserialize(v, target) } -> std::convertible_to<error_code>; }
+                         && requires { field_value.get_value(); }) {
+      // A transparent structure read from a document: the adapter takes an
+      // ondemand::value. A scalar document cannot be viewed as a value, so it
+      // reports SCALAR_DOCUMENT_AS_VALUE (an adapter taking auto& receives the
+      // document itself and has no such limitation).
+      ondemand_value v;
+      SIMDJSON_TRY(field_value.get_value().get(v));
+      return adapter::deserialize(v, target);
+    } else {
+      static_assert(!declares_deserialize(^^adapter),
+                    "the deserialize function of a simdjson::with adapter must be callable as "
+                    "Adapter::deserialize(simdjson::ondemand::value &, T &) and return an error_code");
+      return field_value.get(target);
+    }
+  } else {
+    return field_value.get(target);
+  }
+}
+
+// Deserialize the storage `target` of member `mem` from a JSON value.
+template <auto mem, typename ValueT, typename M>
+simdjson_warn_unused simdjson_inline error_code deserialize_member(ValueT &field_value, M &target) noexcept(false) {
+  if constexpr (has_default<mem>() && std::is_default_constructible_v<M> && std::is_move_assignable_v<M>) {
+    // A present key replaces the default value: deserialize into a fresh
+    // temporary so that, e.g., a container does not append to its default
+    // content, and a failure leaves the default untouched.
+    M value{};
+    SIMDJSON_TRY(deserialize_member_value<mem>(field_value, value));
+    target = std::move(value);
+    return SUCCESS;
+  } else {
+    return deserialize_member_value<mem>(field_value, target);
+  }
+}

+// Deserialize the field with the given field index.
+template <typename T>
+simdjson_warn_unused simdjson_inline error_code deserialize_field_at(
+    std::size_t field_index, haswell::ondemand::value field_value, T &out) noexcept(false) {
+  std::size_t counter = 0;
+  template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+    using field = [: path :];
+    if (field_index == counter) { return deserialize_member<field::leaf>(field_value, field::get(out)); }
+    ++counter;
+  }
+  return SUCCESS;
+}
+
+// Called when the key(s) of member `mem`, stored in `target`, are missing from
+// the JSON object: an error for a required member, a call to the default_from
+// factory, or nothing at all.
+template <auto mem, typename M>
+simdjson_warn_unused simdjson_inline error_code on_missing_member(M &target) noexcept(false) {
+  constexpr std::meta::info default_from_type = simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::default_from_t);
+  if constexpr (default_from_type != std::meta::info{}) {
+    target = [: default_from_type :]::factory();
+    return SUCCESS;
+  } else if constexpr (may_be_absent<mem>()) {
+    // For optional and default_value members, a missing key is not an error:
+    // leave the member at its current (default) value.
+    (void)target;
+    return SUCCESS;
+  } else {
+    (void)target;
+    return NO_SUCH_FIELD;
+  }
+}
+
+// Report or handle every field that was not seen in the JSON object.
+template <typename T, std::size_t N>
+simdjson_warn_unused simdjson_inline error_code handle_missing_fields(
+    const std::array<bool, N> &seen_field, T &out) noexcept(false) {
+  std::size_t counter = 0;
+  template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+    using field = [: path :];
+    if (!seen_field[counter]) { SIMDJSON_TRY(on_missing_member<field::leaf>(field::get(out))); }
+    ++counter;
+  }
+  return SUCCESS;
+}
+
+// Ordered, per-field deserialization: one obj[key] lookup per eligible field
+// (and per alias, until one is found). This is the opt-out path
+// (-DSIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION=1), except for structs with keys
+// that need unescaping (see keys_need_unescaping).
+template <typename T>
+simdjson_warn_unused error_code deserialize_struct_ordered(
+    haswell::ondemand::object &obj, T &out) noexcept(false) {
+  template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+    using field = [: path :];
+    haswell::ondemand::value field_value;
+    error_code error = NO_SUCH_FIELD;
+    template for (constexpr const char *key : std::define_static_array(accepted_keys_of(field::leaf))) {
+      if (error == NO_SUCH_FIELD) { error = obj[std::string_view(key)].get(field_value); }
+    }
+    if (error == NO_SUCH_FIELD) {
+      SIMDJSON_TRY(on_missing_member<field::leaf>(field::get(out)));
+    } else if (error) {
+      return error;
+    } else {
+      SIMDJSON_TRY(deserialize_member<field::leaf>(field_value, field::get(out)));
+    }
+  }
+  return SUCCESS;
+}
+
+// Appends the JSON keys that serialization writes for members of `type` that
+// deserialization cannot assign (const or non-public members, directly or
+// through flatten). `all` is true inside a flattened member that is itself
+// unassignable. Keys of skip_deserializing members are not included: they are
+// unknown keys, as documented.
+consteval void append_unassignable_keys(std::meta::info type, bool all, std::vector<const char *> &keys) {
+  for (std::meta::info mem : std::meta::nonstatic_data_members_of(type, std::meta::access_context::unchecked())) {
+    if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_tag)
+        || simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_serializing_tag)
+        || simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_deserializing_tag)) {
+      continue;
+    }
+    bool unassignable = all || !is_eligible_member(mem);
+    if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::flatten_tag)) {
+      append_unassignable_keys(simdjson::detail::flattened_type(mem), unassignable, keys);
+    } else if (unassignable) {
+      keys.push_back(std::define_static_string(simdjson::detail::json_key_name(mem)));
+    }
+  }
+}
+
+consteval std::vector<const char *> unassignable_keys(std::meta::info type) {
+  std::vector<const char *> keys;
+  append_unassignable_keys(type, false, keys);
+  return keys;
+}
+
+// Deserialization by a single pass over every field of the object, comparing
+// unescaped keys. It is used for structs annotated with deny_unknown_fields
+// (DenyUnknown = true), where a key that does not map to an eligible field is
+// reported as UNKNOWN_FIELD, and as the fallback for structs whose keys the
+// key_selector or obj[key] cannot match (see keys_fit_selector and
+// keys_need_unescaping). As with object::for_each, the first occurrence of a
+// field wins (later duplicates, or aliases of a field already seen, are ignored).
+template <bool DenyUnknown, typename T>
+simdjson_warn_unused error_code deserialize_struct_scan(
+    haswell::ondemand::object &obj, T &out) noexcept(false) {
+  static constexpr auto keys = std::define_static_array(accepted_keys(^^T));
+  static constexpr auto key_fields = std::define_static_array(accepted_key_fields(^^T));
+  std::array<bool, eligible_field_count<T>()> seen_field{};
+  for (auto field_result : obj) {
+    haswell::ondemand::field json_field;
+    SIMDJSON_TRY(std::move(field_result).get(json_field));
+    std::string_view key;
+    SIMDJSON_TRY(json_field.unescaped_key().get(key));
+    std::size_t key_index = keys.size();
+    for (std::size_t i = 0; i < keys.size(); ++i) {
+      if (key == std::string_view(keys[i])) { key_index = i; break; }
+    }
+    if (key_index == keys.size()) {
+      if constexpr (DenyUnknown) {
+        // A key that T itself serializes (e.g. of a const member) is not
+        // unknown: a serialized value must parse back.
+        static constexpr auto ignored_keys = std::define_static_array(unassignable_keys(^^T));
+        bool ignored = false;
+        for (const char *ignored_key : ignored_keys) {
+          if (key == std::string_view(ignored_key)) { ignored = true; break; }
+        }
+        if (!ignored) { return UNKNOWN_FIELD; }
+      }
+      continue;
+    }
+    const std::size_t field_index = key_fields[key_index];
+    if (seen_field[field_index]) { continue; }
+    seen_field[field_index] = true;
+    SIMDJSON_TRY(deserialize_field_at(field_index, json_field.value(), out));
+  }
+  return handle_missing_fields(seen_field, out);
+}
+
+template <typename T>
+consteval bool is_transparent() {
+  return simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag);
+}
+
+} // namespace key_selector_reflection_detail
+
+// Deserialize a reflected struct. By default this builds a compile-time
+// key_selector from the struct's members and walks each object once with
+// object::for_each (perfect-hash key matching), instead of one obj[key] lookup
+// per member. Other paths are used instead:
+//   - globally, the ordered per-member path when defining
+//     -DSIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION=1;
+//   - automatically and per-type, a scan of the object comparing unescaped keys
+//     when the struct's keys do not fit the key_selector limits (see
+//     keys_fit_selector) or cannot be compared raw (see keys_need_unescaping),
+//     so that long member names and the like keep compiling rather than
+//     tripping a static_assert.
+// Structs annotated with deny_unknown_fields always use the strict scan, and
+// structs annotated with transparent are deserialized as their single member.
+//
+// noexcept(false): a member's tag_invoke may throw and the exception must reach
+// the caller of get<T>().
 template <typename T, typename ValT>
   requires(user_defined_type<T> && std::is_class_v<T>)
-error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
+  if constexpr (key_selector_reflection_detail::is_transparent<T>()) {
+    constexpr auto mem = simdjson::detail::transparent_member(^^T);
+    if constexpr (std::is_same_v<std::remove_cvref_t<ValT>, haswell::ondemand::object>) {
+      // We were handed an object: only a structure can be deserialized from it.
+      if constexpr (user_defined_type<typename [: std::meta::type_of(mem) :]>) {
+        return tag_invoke(deserialize_tag{}, val, out.[:mem:]);
+      } else {
+        return INCORRECT_TYPE;
+      }
+    } else {
+      return key_selector_reflection_detail::deserialize_member<mem>(val, out.[:mem:]);
+    }
+  } else {
+  static_assert(key_selector_reflection_detail::accepted_keys_are_distinct<T>(),
+                "two members of this structure accept the same JSON key (check rename, alias, "
+                "rename_all and flatten)");
   haswell::ondemand::object obj;
   if constexpr (std::is_same_v<std::remove_cvref_t<ValT>, haswell::ondemand::object>) {
     obj = val;
   } else {
     SIMDJSON_TRY(val.get_object().get(obj));
   }
-  template for (constexpr auto mem : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
-    if constexpr (!std::meta::is_const(mem) && std::meta::is_public(mem)) {
-      constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
-      if constexpr (concepts::optional_type<decltype(out.[:mem:])>) {
-        // for optional members, it's ok if the key is missing
-        auto error = obj[key].get(out.[:mem:]);
-        if (error && error != NO_SUCH_FIELD) {
-          if(error == NO_SUCH_FIELD) {
-            out.[:mem:].reset();
-            continue;
-          }
-          return error;
-        }
-      } else {
-        // for non-optional members, the key must be present
-        SIMDJSON_TRY(obj[key].get(out.[:mem:]));
+  if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::deny_unknown_fields_tag)) {
+    return key_selector_reflection_detail::deserialize_struct_scan<true>(obj, out);
+  } else {
+#if defined(SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION) && SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION
+  // Opt-out build: use the ordered per-member path, unless obj[key] cannot
+  // match T's keys.
+  if constexpr (key_selector_reflection_detail::keys_need_unescaping<T>()) {
+    return key_selector_reflection_detail::deserialize_struct_scan<false>(obj, out);
+  } else {
+    return key_selector_reflection_detail::deserialize_struct_ordered(obj, out);
+  }
+#else
+  if constexpr (key_selector_reflection_detail::eligible_field_count<T>() == 0) {
+    // No fields to deserialize: an empty key_selector cannot be built, so just
+    // validate that the input is an object (done above) and succeed. Mirrors the
+    // ordered per-member path, which iterates over zero members.
+    (void)out;
+    (void)obj;
+    return SUCCESS;
+  } else if constexpr (!key_selector_reflection_detail::keys_fit_selector<T>()) {
+    // Automatic fallback: T's accepted keys do not fit the key_selector limits
+    // (e.g. a member name longer than 63 characters, or a key with a double
+    // quote), so building a selector would be a compile error. Scan the object
+    // instead, so the default never breaks a struct that the opt-out path would
+    // accept.
+    return key_selector_reflection_detail::deserialize_struct_scan<false>(obj, out);
+  } else {
+  using selector = key_selector_reflection_detail::selector_for<T>;
+  if constexpr (key_selector_reflection_detail::all_eligible_fields_required<T>()
+                && !key_selector_reflection_detail::has_aliases<T>()) {
+    // Fast path: every member is required and has a single key. A single
+    // for_each pass parses each matched field; the returned match count then
+    // tells us whether every member was present (matched_count ==
+    // selector::size()) without a per-member "seen" array. A value-parse error
+    // (e.g. a type mismatch) is propagated by for_each.
+    auto walk = obj.template for_each<selector>(
+        [&](std::size_t matched_index, haswell::ondemand::value field_value) -> error_code {
+      std::size_t counter = 0;
+      template for (constexpr auto path : std::define_static_array(key_selector_reflection_detail::eligible_fields(^^T))) {
+        using field = [: path :];
+        if (matched_index == counter) { return key_selector_reflection_detail::deserialize_member<field::leaf>(field_value, field::get(out)); }
+        ++counter;
       }
-    }
-  };
-  return simdjson::SUCCESS;
-}
-
-// Support for enum deserialization - deserialize from string representation using expand approach from P2996R12
+      return SUCCESS;
+    });
+    if (walk.error) { return walk.error; }
+    // A missing required member shows up as a short match count and is reported as
+    // NO_SUCH_FIELD, mirroring the ordered obj[key] path.
+    if (walk.matched_count != selector::size()) { return NO_SUCH_FIELD; }
+    return SUCCESS;
+  } else {
+    static constexpr auto key_fields = std::define_static_array(key_selector_reflection_detail::accepted_key_fields(^^T));
+    std::array<bool, key_selector_reflection_detail::eligible_field_count<T>()> seen_field{};
+    // Single pass over the object: each field whose key matches a member (or one
+    // of its aliases) yields its selector index, which we map back to the
+    // corresponding member. The first key seen for a member wins. The callback
+    // returns an error_code so that a value-parse error (e.g. a type mismatch on
+    // a matched field) is propagated by for_each instead of being silently dropped.
+    error_code walk_error = obj.template for_each<selector>(
+        [&](std::size_t matched_index, haswell::ondemand::value field_value) -> error_code {
+      const std::size_t field_index = key_fields[matched_index];
+      if (seen_field[field_index]) { return SUCCESS; }
+      seen_field[field_index] = true;
+      return key_selector_reflection_detail::deserialize_field_at(field_index, field_value, out);
+    });
+    if (walk_error) { return walk_error; }
+    // Required members must be present: a missing one is reported as
+    // NO_SUCH_FIELD, mirroring the ordered obj[key] path. Optional and defaulted
+    // members may be absent.
+    return key_selector_reflection_detail::handle_missing_fields(seen_field, out);
+  }
+  }
+#endif // SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION
+  }
+  }
+}
+
+// Support for enum deserialization - deserialize from string representation using
+// expand approach from P2996R12. The accepted strings are the enumerator's JSON key
+// (see rename and rename_all) and its aliases.
 template <typename T, typename ValT>
   requires(std::is_enum_v<T>)
 error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
 #if SIMDJSON_STATIC_REFLECTION
   std::string_view str;
   SIMDJSON_TRY(val.get_string().get(str));
-  constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
+  static constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
   template for (constexpr auto enum_val : enumerators) {
-    if (str == std::meta::identifier_of(enum_val)) {
-      out = [:enum_val:];
-      return SUCCESS;
+    template for (constexpr const char *key : std::define_static_array(key_selector_reflection_detail::accepted_keys_of(enum_val))) {
+      if (str == std::string_view(key)) {
+        out = [:enum_val:];
+        return SUCCESS;
+      }
     }
   };

@@ -95574,33 +121233,25 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {

 template <typename simdjson_value, typename T>
   requires(user_defined_type<std::remove_cvref_t<T>>)
-error_code tag_invoke(deserialize_tag, simdjson_value &val, std::unique_ptr<T> &out) noexcept {
-  if (!out) {
-    out = std::make_unique<T>();
-    if (!out) {
-      return MEMALLOC;
-    }
-  }
-  if (auto err = val.get(*out)) {
-    out.reset();
-    return err;
+error_code tag_invoke(deserialize_tag, simdjson_value &val, std::unique_ptr<T> &out) noexcept(false) {
+  std::unique_ptr<T> ptr(new (std::nothrow) T());
+  if (!ptr) {
+    return MEMALLOC;
   }
+  SIMDJSON_TRY(val.get(*ptr));
+  out = std::move(ptr);
   return SUCCESS;
 }

 template <typename simdjson_value, typename T>
   requires(user_defined_type<std::remove_cvref_t<T>>)
-error_code tag_invoke(deserialize_tag, simdjson_value &val, std::shared_ptr<T> &out) noexcept {
-  if (!out) {
-    out = std::make_shared<T>();
-    if (!out) {
-      return MEMALLOC;
-    }
-  }
-  if (auto err = val.get(*out)) {
-    out.reset();
-    return err;
+error_code tag_invoke(deserialize_tag, simdjson_value &val, std::shared_ptr<T> &out) noexcept(false) {
+  std::shared_ptr<T> ptr(new (std::nothrow) T());
+  if (!ptr) {
+    return MEMALLOC;
   }
+  SIMDJSON_TRY(val.get(*ptr));
+  out = std::move(ptr);
   return SUCCESS;
 }

@@ -95912,9 +121563,17 @@ simdjson_inline simdjson_result<array> array::started(value_iterator &iter) noex
   return array(iter);
 }

-simdjson_inline simdjson_result<array_iterator> array::begin() noexcept {
+simdjson_inline simdjson_result<array_iterator> array::begin() & noexcept {
 #if SIMDJSON_DEVELOPMENT_CHECKS
-  if (!iter.is_at_iterator_start()) { return OUT_OF_ORDER_ITERATION; }
+  if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
+  return array_iterator(iter, this);
+#endif
+  return array_iterator(iter);
+}
+simdjson_inline simdjson_result<array_iterator> array::begin() && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  // The array is a temporary that the iterator may outlive: do not lock it.
+  if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
 #endif
   return array_iterator(iter);
 }
@@ -95941,6 +121600,9 @@ simdjson_inline simdjson_result<std::string_view> array::raw_json() noexcept {
 SIMDJSON_PUSH_DISABLE_WARNINGS
 SIMDJSON_DISABLE_STRICT_OVERFLOW_WARNING
 simdjson_inline simdjson_result<size_t> array::count_elements() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   size_t count{0};
   // Important: we do not consume any of the values.
   for(simdjson_unused auto v : *this) { count++; }
@@ -95954,6 +121616,9 @@ simdjson_inline simdjson_result<size_t> array::count_elements() & noexcept {
 SIMDJSON_POP_DISABLE_WARNINGS

 simdjson_inline simdjson_result<bool> array::is_empty() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   bool is_not_empty;
   auto error = iter.reset_array().get(is_not_empty);
   if(error) { return error; }
@@ -95961,31 +121626,30 @@ simdjson_inline simdjson_result<bool> array::is_empty() & noexcept {
 }

 inline simdjson_result<bool> array::reset() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   return iter.reset_array();
 }

+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline void array::set_locked(bool _locked) noexcept {
+  locked = _locked;
+}
+#endif
+
 inline simdjson_result<value> array::at_pointer(std::string_view json_pointer) noexcept {
-  if (json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
+  // An empty pointer has no json_pointer[0]: with a default-constructed
+  // std::string_view, reading it dereferences a null pointer.
+  if (json_pointer.empty() || json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
   json_pointer = json_pointer.substr(1);
   // - means "the append position" or "the element after the end of the array"
   // We don't support this, because we're returning a real element, not a position.
   if (json_pointer == "-") { return INDEX_OUT_OF_BOUNDS; }

-  // Read the array index
   size_t array_index = 0;
   size_t i;
-  for (i = 0; i < json_pointer.length() && json_pointer[i] != '/'; i++) {
-    uint8_t digit = uint8_t(json_pointer[i] - '0');
-    // Check for non-digit in array index. If it's there, we're trying to get a field in an object
-    if (digit > 9) { return INCORRECT_TYPE; }
-    array_index = array_index*10 + digit;
-  }
-
-  // 0 followed by other digits is invalid
-  if (i > 1 && json_pointer[0] == '0') { return INVALID_JSON_POINTER; } // "JSON pointer array index has other characters after 0"
-
-  // Empty string is invalid; so is a "/" with no digits before it
-  if (i == 0) { return INVALID_JSON_POINTER; } // "Empty string in JSON pointer array index"
+  SIMDJSON_TRY(internal::parse_json_pointer_array_index(json_pointer, array_index, i));
   // Get the child
   auto child = at(array_index);
   // If there is an error, it ends here
@@ -96059,6 +121723,9 @@ inline error_code array::for_each_at_path_with_wildcard(std::string_view json_pa
 }

 simdjson_inline simdjson_result<value> array::at(size_t index) noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   size_t i = 0;
   for (auto value : *this) {
     if (i == index) { return value; }
@@ -96088,10 +121755,14 @@ simdjson_inline simdjson_result<haswell::ondemand::array>::simdjson_result(
 {
 }

-simdjson_inline simdjson_result<haswell::ondemand::array_iterator> simdjson_result<haswell::ondemand::array>::begin() noexcept {
+simdjson_inline simdjson_result<haswell::ondemand::array_iterator> simdjson_result<haswell::ondemand::array>::begin() & noexcept {
   if (error()) { return error(); }
   return first.begin();
 }
+simdjson_inline simdjson_result<haswell::ondemand::array_iterator> simdjson_result<haswell::ondemand::array>::begin() && noexcept {
+  if (error()) { return error(); }
+  return std::move(first).begin();
+}
 simdjson_inline simdjson_result<haswell::ondemand::array_iterator> simdjson_result<haswell::ondemand::array>::end() noexcept {
   if (error()) { return error(); }
   return first.end();
@@ -96154,6 +121825,59 @@ simdjson_inline array_iterator::array_iterator(const value_iterator &_iter) noex
   : iter{_iter}
 {}

+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline array_iterator::array_iterator(const value_iterator &_iter, array* _parent) noexcept
+  : parent{_parent}, iter{_iter}
+{
+  if (parent) parent->set_locked(true);
+}
+
+simdjson_inline array_iterator::~array_iterator() noexcept
+{
+  if (parent) parent->set_locked(false);
+}
+
+simdjson_inline array_iterator::array_iterator(array_iterator&& other) noexcept
+  : has_been_referenced{other.has_been_referenced},
+    parent{other.parent},
+    iter{std::move(other.iter)}
+{
+  other.parent = nullptr;
+}
+
+simdjson_inline array_iterator& array_iterator::operator=(array_iterator&& other) noexcept {
+  if (this != &other)
+  {
+    if (parent)
+      parent->set_locked(false);
+    has_been_referenced = other.has_been_referenced;
+    parent = other.parent;
+    iter = std::move(other.iter);
+
+    other.parent = nullptr;
+  }
+  return *this;
+}
+
+simdjson_inline array_iterator::array_iterator(const array_iterator& other) noexcept
+  : has_been_referenced{other.has_been_referenced},
+    parent{nullptr},
+    iter{other.iter}
+{}
+
+simdjson_inline array_iterator& array_iterator::operator=(const array_iterator& other) noexcept {
+  if (this != &other)
+  {
+    if (parent)
+      parent->set_locked(false);
+    has_been_referenced = other.has_been_referenced;
+    parent = nullptr;
+    iter = other.iter;
+  }
+  return *this;
+}
+#endif
+
 simdjson_inline simdjson_result<value> array_iterator::operator*() noexcept {
 #if SIMDJSON_DEVELOPMENT_CHECKS
    SIMDJSON_ASSUME(!has_been_referenced);
@@ -96249,6 +121973,41 @@ namespace simdjson {
 namespace haswell {
 namespace ondemand {

+/**
+ * @private Narrows the result of get_uint64() to a smaller unsigned integer type T.
+ * @returns NUMBER_OUT_OF_RANGE if the value does not fit in a T.
+ */
+template <typename T>
+simdjson_inline simdjson_result<T> narrow_integer(simdjson_result<uint64_t> wide) noexcept {
+  uint64_t result;
+  SIMDJSON_TRY(std::move(wide).get(result));
+  if (result > static_cast<uint64_t>((std::numeric_limits<T>::max)())) { return NUMBER_OUT_OF_RANGE; }
+  return static_cast<T>(result);
+}
+
+/**
+ * @private Narrows the result of get_int64() to a smaller signed integer type T.
+ * @returns NUMBER_OUT_OF_RANGE if the value does not fit in a T.
+ */
+template <typename T>
+simdjson_inline simdjson_result<T> narrow_integer(simdjson_result<int64_t> wide) noexcept {
+  int64_t result;
+  SIMDJSON_TRY(std::move(wide).get(result));
+  if (result > static_cast<int64_t>((std::numeric_limits<T>::max)()) || result < static_cast<int64_t>((std::numeric_limits<T>::min)())) { return NUMBER_OUT_OF_RANGE; }
+  return static_cast<T>(result);
+}
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+// get_float32() parses with get_float(), so float must be binary32.
+static_assert(std::numeric_limits<float>::is_iec559 && std::numeric_limits<float>::digits == 24,
+              "get_float32() requires float to be IEEE binary32");
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+// get_float64() parses with get_double(), so double must be binary64.
+static_assert(std::numeric_limits<double>::is_iec559 && std::numeric_limits<double>::digits == 53,
+              "get_float64() requires double to be IEEE binary64");
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
+
 simdjson_inline value::value(const value_iterator &_iter) noexcept
   : iter{_iter}
 {
@@ -96280,6 +122039,13 @@ simdjson_inline simdjson_result<raw_json_string> value::get_raw_json_string() no
 simdjson_inline simdjson_result<std::string_view> value::get_string(bool allow_replacement) noexcept {
   return iter.get_string(allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> value::get_u8string(bool allow_replacement) noexcept {
+  std::string_view content;
+  SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+  return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code value::get_string(string_type& receiver, bool allow_replacement) noexcept {
   return iter.get_string(receiver, allow_replacement);
@@ -96293,6 +122059,12 @@ simdjson_inline simdjson_result<double> value::get_double() noexcept {
 simdjson_inline simdjson_result<double> value::get_double_in_string() noexcept {
   return iter.get_double_in_string();
 }
+simdjson_inline simdjson_result<float> value::get_float() noexcept {
+  return iter.get_float();
+}
+simdjson_inline simdjson_result<float> value::get_float_in_string() noexcept {
+  return iter.get_float_in_string();
+}
 simdjson_inline simdjson_result<uint64_t> value::get_uint64() noexcept {
   return iter.get_uint64();
 }
@@ -96306,17 +122078,37 @@ simdjson_inline simdjson_result<int64_t> value::get_int64_in_string() noexcept {
   return iter.get_int64_in_string();
 }
 simdjson_inline simdjson_result<uint32_t> value::get_uint32() noexcept {
-  uint64_t result;
-  SIMDJSON_TRY(get_uint64().get(result));
-  if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<uint32_t>(result);
+  return narrow_integer<uint32_t>(get_uint64());
 }
 simdjson_inline simdjson_result<int32_t> value::get_int32() noexcept {
-  int64_t result;
-  SIMDJSON_TRY(get_int64().get(result));
-  if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<int32_t>(result);
+  return narrow_integer<int32_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint16_t> value::get_uint16() noexcept {
+  return narrow_integer<uint16_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int16_t> value::get_int16() noexcept {
+  return narrow_integer<int16_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint8_t> value::get_uint8() noexcept {
+  return narrow_integer<uint8_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int8_t> value::get_int8() noexcept {
+  return narrow_integer<int8_t>(get_int64());
 }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> value::get_float32() noexcept {
+  float result;
+  SIMDJSON_TRY(get_float().get(result));
+  return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> value::get_float64() noexcept {
+  double result;
+  SIMDJSON_TRY(get_double().get(result));
+  return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<bool> value::get_bool() noexcept {
   return iter.get_bool();
 }
@@ -96328,12 +122120,26 @@ template<> simdjson_inline simdjson_result<array> value::get() noexcept { return
 template<> simdjson_inline simdjson_result<object> value::get() noexcept { return get_object(); }
 template<> simdjson_inline simdjson_result<raw_json_string> value::get() noexcept { return get_raw_json_string(); }
 template<> simdjson_inline simdjson_result<std::string_view> value::get() noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> value::get() noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_inline simdjson_result<number> value::get() noexcept { return get_number(); }
 template<> simdjson_inline simdjson_result<double> value::get() noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> value::get() noexcept { return get_float(); }
 template<> simdjson_inline simdjson_result<uint64_t> value::get() noexcept { return get_uint64(); }
 template<> simdjson_inline simdjson_result<int64_t> value::get() noexcept { return get_int64(); }
 template<> simdjson_inline simdjson_result<uint32_t> value::get() noexcept { return get_uint32(); }
 template<> simdjson_inline simdjson_result<int32_t> value::get() noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> value::get() noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> value::get() noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> value::get() noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> value::get() noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> value::get() noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> value::get() noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<> simdjson_inline simdjson_result<bool> value::get() noexcept { return get_bool(); }


@@ -96341,12 +122147,26 @@ template<> simdjson_warn_unused simdjson_inline error_code value::get(array& out
 template<> simdjson_warn_unused simdjson_inline error_code value::get(object& out) noexcept { return get_object().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(raw_json_string& out) noexcept { return get_raw_json_string().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(std::string_view& out) noexcept { return get_string(false).get(out); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::u8string_view& out) noexcept { return get_u8string(false).get(out); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_warn_unused simdjson_inline error_code value::get(number& out) noexcept { return get_number().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(double& out) noexcept { return get_double().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(float& out) noexcept { return get_float().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(uint64_t& out) noexcept { return get_uint64().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(int64_t& out) noexcept { return get_int64().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(uint32_t& out) noexcept { return get_uint32().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(int32_t& out) noexcept { return get_int32().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(uint16_t& out) noexcept { return get_uint16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(int16_t& out) noexcept { return get_int16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(uint8_t& out) noexcept { return get_uint8().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(int8_t& out) noexcept { return get_int8().get(out); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::float32_t& out) noexcept { return get_float32().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::float64_t& out) noexcept { return get_float64().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<>  simdjson_warn_unused simdjson_inline error_code value::get(bool& out) noexcept { return get_bool().get(out); }

 #if SIMDJSON_EXCEPTIONS
@@ -96515,6 +122335,9 @@ inline bool is_pointer_well_formed(std::string_view json_pointer) noexcept {
 }

 simdjson_inline simdjson_result<value> value::at_pointer(std::string_view json_pointer) noexcept {
+  // The empty JSON Pointer refers to the whole value (RFC 6901), as in
+  // document::at_pointer.
+  if (json_pointer.empty()) { return value(iter); }
   json_type t;
   SIMDJSON_TRY(type().get(t));
   switch (t)
@@ -96552,6 +122375,10 @@ template <typename Func>
 template <typename Func>
 #endif
 inline error_code value::for_each_at_path_with_wildcard(std::string_view json_path, Func&& callback) noexcept {
+  // Every recursive step of for_each_at_path_with_wildcard goes through this
+  // function, and each one descends one level into the document. A path with
+  // many segments applied to a deeply nested document would otherwise recurse
+  // without bound and overflow the stack.
   if (size_t(iter.depth()) >= iter.json_iter().parser->max_depth()) { return DEPTH_ERROR; }
   json_type t;
   SIMDJSON_TRY(type().get(t));
@@ -96665,10 +122492,46 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<haswell::ondemand::valu
   if (error()) { return error(); }
   return first.get_int32();
 }
+simdjson_inline simdjson_result<uint16_t> simdjson_result<haswell::ondemand::value>::get_uint16() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<haswell::ondemand::value>::get_int16() noexcept {
+  if (error()) { return error(); }
+  return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<haswell::ondemand::value>::get_uint8() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<haswell::ondemand::value>::get_int8() noexcept {
+  if (error()) { return error(); }
+  return first.get_int8();
+}
 simdjson_inline simdjson_result<double> simdjson_result<haswell::ondemand::value>::get_double() noexcept {
   if (error()) { return error(); }
   return first.get_double();
 }
+simdjson_inline simdjson_result<float> simdjson_result<haswell::ondemand::value>::get_float() noexcept {
+  if (error()) { return error(); }
+  return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<haswell::ondemand::value>::get_float_in_string() noexcept {
+  if (error()) { return error(); }
+  return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<haswell::ondemand::value>::get_float32() noexcept {
+  if (error()) { return error(); }
+  return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<haswell::ondemand::value>::get_float64() noexcept {
+  if (error()) { return error(); }
+  return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<double> simdjson_result<haswell::ondemand::value>::get_double_in_string() noexcept {
   if (error()) { return error(); }
   return first.get_double_in_string();
@@ -96677,6 +122540,12 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<haswell::ondem
   if (error()) { return error(); }
   return first.get_string(allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<haswell::ondemand::value>::get_u8string(bool allow_replacement) noexcept {
+  if (error()) { return error(); }
+  return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_inline error_code simdjson_result<haswell::ondemand::value>::get_string(string_type& receiver, bool allow_replacement) noexcept {
   if (error()) { return error(); }
@@ -96705,11 +122574,23 @@ template<> simdjson_inline error_code simdjson_result<haswell::ondemand::value>:
   return SUCCESS;
 }

-template<typename T> simdjson_inline simdjson_result<T> simdjson_result<haswell::ondemand::value>::get() noexcept {
+template<typename T> simdjson_inline simdjson_result<T> simdjson_result<haswell::ondemand::value>::get()
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, haswell::ondemand::value>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>();
 }
-template<typename T> simdjson_inline error_code simdjson_result<haswell::ondemand::value>::get(T &out) noexcept {
+template<typename T> simdjson_inline error_code simdjson_result<haswell::ondemand::value>::get(T &out)
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, haswell::ondemand::value>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>(out);
 }
@@ -96979,16 +122860,22 @@ simdjson_inline simdjson_result<int64_t> document::get_int64_in_string() noexcep
   return get_root_value_iterator().get_root_int64_in_string(true);
 }
 simdjson_inline simdjson_result<uint32_t> document::get_uint32() noexcept {
-  uint64_t result;
-  SIMDJSON_TRY(get_uint64().get(result));
-  if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<uint32_t>(result);
+  return narrow_integer<uint32_t>(get_uint64());
 }
 simdjson_inline simdjson_result<int32_t> document::get_int32() noexcept {
-  int64_t result;
-  SIMDJSON_TRY(get_int64().get(result));
-  if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<int32_t>(result);
+  return narrow_integer<int32_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint16_t> document::get_uint16() noexcept {
+  return narrow_integer<uint16_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int16_t> document::get_int16() noexcept {
+  return narrow_integer<int16_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint8_t> document::get_uint8() noexcept {
+  return narrow_integer<uint8_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int8_t> document::get_int8() noexcept {
+  return narrow_integer<int8_t>(get_int64());
 }
 simdjson_inline simdjson_result<double> document::get_double() noexcept {
   return get_root_value_iterator().get_root_double(true);
@@ -96996,9 +122883,36 @@ simdjson_inline simdjson_result<double> document::get_double() noexcept {
 simdjson_inline simdjson_result<double> document::get_double_in_string() noexcept {
   return get_root_value_iterator().get_root_double_in_string(true);
 }
+simdjson_inline simdjson_result<float> document::get_float() noexcept {
+  return get_root_value_iterator().get_root_float(true);
+}
+simdjson_inline simdjson_result<float> document::get_float_in_string() noexcept {
+  return get_root_value_iterator().get_root_float_in_string(true);
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> document::get_float32() noexcept {
+  float result;
+  SIMDJSON_TRY(get_float().get(result));
+  return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> document::get_float64() noexcept {
+  double result;
+  SIMDJSON_TRY(get_double().get(result));
+  return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<std::string_view> document::get_string(bool allow_replacement) noexcept {
   return get_root_value_iterator().get_root_string(true, allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> document::get_u8string(bool allow_replacement) noexcept {
+  std::string_view content;
+  SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+  return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code document::get_string(string_type& receiver, bool allow_replacement) noexcept {
   return get_root_value_iterator().get_root_string(receiver, true, allow_replacement);
@@ -97020,11 +122934,25 @@ template<> simdjson_inline simdjson_result<array> document::get() & noexcept { r
 template<> simdjson_inline simdjson_result<object> document::get() & noexcept { return get_object(); }
 template<> simdjson_inline simdjson_result<raw_json_string> document::get() & noexcept { return get_raw_json_string(); }
 template<> simdjson_inline simdjson_result<std::string_view> document::get() & noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> document::get() & noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_inline simdjson_result<double> document::get() & noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> document::get() & noexcept { return get_float(); }
 template<> simdjson_inline simdjson_result<uint64_t> document::get() & noexcept { return get_uint64(); }
 template<> simdjson_inline simdjson_result<int64_t> document::get() & noexcept { return get_int64(); }
 template<> simdjson_inline simdjson_result<uint32_t> document::get() & noexcept { return get_uint32(); }
 template<> simdjson_inline simdjson_result<int32_t> document::get() & noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> document::get() & noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> document::get() & noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> document::get() & noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> document::get() & noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> document::get() & noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> document::get() & noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<> simdjson_inline simdjson_result<bool> document::get() & noexcept { return get_bool(); }
 template<> simdjson_inline simdjson_result<value> document::get() & noexcept { return get_value(); }

@@ -97032,17 +122960,35 @@ template<> simdjson_warn_unused simdjson_inline error_code document::get(array&
 template<> simdjson_warn_unused simdjson_inline error_code document::get(object& out) & noexcept { return get_object().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(raw_json_string& out) & noexcept { return get_raw_json_string().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(std::string_view& out) & noexcept { return get_string(false).get(out); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::u8string_view& out) & noexcept { return get_u8string(false).get(out); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_warn_unused simdjson_inline error_code document::get(double& out) & noexcept { return get_double().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(float& out) & noexcept { return get_float().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(uint64_t& out) & noexcept { return get_uint64().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(int64_t& out) & noexcept { return get_int64().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(uint32_t& out) & noexcept { return get_uint32().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(int32_t& out) & noexcept { return get_int32().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(uint16_t& out) & noexcept { return get_uint16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(int16_t& out) & noexcept { return get_int16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(uint8_t& out) & noexcept { return get_uint8().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(int8_t& out) & noexcept { return get_int8().get(out); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::float32_t& out) & noexcept { return get_float32().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::float64_t& out) & noexcept { return get_float64().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<> simdjson_warn_unused simdjson_inline error_code document::get(bool& out) & noexcept { return get_bool().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(value& out) & noexcept { return get_value().get(out); }

 template<> simdjson_deprecated simdjson_inline simdjson_result<raw_json_string> document::get() && noexcept { return get_raw_json_string(); }
 template<> simdjson_deprecated simdjson_inline simdjson_result<std::string_view> document::get() && noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_deprecated simdjson_inline simdjson_result<std::u8string_view> document::get() && noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_deprecated simdjson_inline simdjson_result<double> document::get() && noexcept { return std::forward<document>(*this).get_double(); }
+template<> simdjson_deprecated simdjson_inline simdjson_result<float> document::get() && noexcept { return std::forward<document>(*this).get_float(); }
 template<> simdjson_deprecated simdjson_inline simdjson_result<uint64_t> document::get() && noexcept { return std::forward<document>(*this).get_uint64(); }
 template<> simdjson_deprecated simdjson_inline simdjson_result<int64_t> document::get() && noexcept { return std::forward<document>(*this).get_int64(); }
 template<> simdjson_deprecated simdjson_inline simdjson_result<bool> document::get() && noexcept { return std::forward<document>(*this).get_bool(); }
@@ -97381,6 +123327,22 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<haswell::ondemand::docu
   if (error()) { return error(); }
   return first.get_int32();
 }
+simdjson_inline simdjson_result<uint16_t> simdjson_result<haswell::ondemand::document>::get_uint16() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<haswell::ondemand::document>::get_int16() noexcept {
+  if (error()) { return error(); }
+  return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<haswell::ondemand::document>::get_uint8() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<haswell::ondemand::document>::get_int8() noexcept {
+  if (error()) { return error(); }
+  return first.get_int8();
+}
 simdjson_inline simdjson_result<double> simdjson_result<haswell::ondemand::document>::get_double() noexcept {
   if (error()) { return error(); }
   return first.get_double();
@@ -97389,10 +123351,36 @@ simdjson_inline simdjson_result<double> simdjson_result<haswell::ondemand::docum
   if (error()) { return error(); }
   return first.get_double_in_string();
 }
+simdjson_inline simdjson_result<float> simdjson_result<haswell::ondemand::document>::get_float() noexcept {
+  if (error()) { return error(); }
+  return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<haswell::ondemand::document>::get_float_in_string() noexcept {
+  if (error()) { return error(); }
+  return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<haswell::ondemand::document>::get_float32() noexcept {
+  if (error()) { return error(); }
+  return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<haswell::ondemand::document>::get_float64() noexcept {
+  if (error()) { return error(); }
+  return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<std::string_view> simdjson_result<haswell::ondemand::document>::get_string(bool allow_replacement) noexcept {
   if (error()) { return error(); }
   return first.get_string(allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<haswell::ondemand::document>::get_u8string(bool allow_replacement) noexcept {
+  if (error()) { return error(); }
+  return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code simdjson_result<haswell::ondemand::document>::get_string(string_type& receiver, bool allow_replacement) noexcept {
   if (error()) { return error(); }
@@ -97420,22 +123408,46 @@ simdjson_inline simdjson_result<bool> simdjson_result<haswell::ondemand::documen
 }

 template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<haswell::ondemand::document>::get() & noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<haswell::ondemand::document>::get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, haswell::ondemand::document>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>();
 }
 template<typename T>
-simdjson_deprecated simdjson_inline simdjson_result<T> simdjson_result<haswell::ondemand::document>::get() && noexcept {
+simdjson_deprecated simdjson_inline simdjson_result<T> simdjson_result<haswell::ondemand::document>::get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, haswell::ondemand::document>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return std::forward<haswell::ondemand::document>(first).get<T>();
 }
 template<typename T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<haswell::ondemand::document>::get(T &out) & noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<haswell::ondemand::document>::get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, haswell::ondemand::document>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>(out);
 }
 template<typename T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<haswell::ondemand::document>::get(T &out) && noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<haswell::ondemand::document>::get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, haswell::ondemand::document>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return std::forward<haswell::ondemand::document>(first).get<T>(out);
 }
@@ -97504,27 +123516,27 @@ simdjson_inline simdjson_result<haswell::ondemand::document>::operator haswell::
 }
 simdjson_inline simdjson_result<haswell::ondemand::document>::operator uint64_t() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_uint64();
 }
 simdjson_inline simdjson_result<haswell::ondemand::document>::operator int64_t() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_int64();
 }
 simdjson_inline simdjson_result<haswell::ondemand::document>::operator double() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_double();
 }
 simdjson_inline simdjson_result<haswell::ondemand::document>::operator std::string_view() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_string();
 }
 simdjson_inline simdjson_result<haswell::ondemand::document>::operator haswell::ondemand::raw_json_string() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_raw_json_string();
 }
 simdjson_inline simdjson_result<haswell::ondemand::document>::operator bool() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_bool();
 }
 simdjson_inline simdjson_result<haswell::ondemand::document>::operator haswell::ondemand::value() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
@@ -97614,21 +123626,38 @@ simdjson_inline simdjson_result<uint64_t> document_reference::get_uint64() noexc
 simdjson_inline simdjson_result<uint64_t> document_reference::get_uint64_in_string() noexcept { return doc->get_root_value_iterator().get_root_uint64_in_string(false); }
 simdjson_inline simdjson_result<int64_t> document_reference::get_int64() noexcept { return doc->get_root_value_iterator().get_root_int64(false); }
 simdjson_inline simdjson_result<int64_t> document_reference::get_int64_in_string() noexcept { return doc->get_root_value_iterator().get_root_int64_in_string(false); }
-simdjson_inline simdjson_result<uint32_t> document_reference::get_uint32() noexcept {
-  uint64_t result;
-  SIMDJSON_TRY(doc->get_root_value_iterator().get_root_uint64(false).get(result));
-  if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<uint32_t>(result);
-}
-simdjson_inline simdjson_result<int32_t> document_reference::get_int32() noexcept {
-  int64_t result;
-  SIMDJSON_TRY(doc->get_root_value_iterator().get_root_int64(false).get(result));
-  if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<int32_t>(result);
-}
+simdjson_inline simdjson_result<uint32_t> document_reference::get_uint32() noexcept { return narrow_integer<uint32_t>(get_uint64()); }
+simdjson_inline simdjson_result<int32_t> document_reference::get_int32() noexcept { return narrow_integer<int32_t>(get_int64()); }
+simdjson_inline simdjson_result<uint16_t> document_reference::get_uint16() noexcept { return narrow_integer<uint16_t>(get_uint64()); }
+simdjson_inline simdjson_result<int16_t> document_reference::get_int16() noexcept { return narrow_integer<int16_t>(get_int64()); }
+simdjson_inline simdjson_result<uint8_t> document_reference::get_uint8() noexcept { return narrow_integer<uint8_t>(get_uint64()); }
+simdjson_inline simdjson_result<int8_t> document_reference::get_int8() noexcept { return narrow_integer<int8_t>(get_int64()); }
 simdjson_inline simdjson_result<double> document_reference::get_double() noexcept { return doc->get_root_value_iterator().get_root_double(false); }
-simdjson_inline simdjson_result<double> document_reference::get_double_in_string() noexcept { return doc->get_root_value_iterator().get_root_double(false); }
+simdjson_inline simdjson_result<double> document_reference::get_double_in_string() noexcept { return doc->get_root_value_iterator().get_root_double_in_string(false); }
+simdjson_inline simdjson_result<float> document_reference::get_float() noexcept { return doc->get_root_value_iterator().get_root_float(false); }
+simdjson_inline simdjson_result<float> document_reference::get_float_in_string() noexcept { return doc->get_root_value_iterator().get_root_float_in_string(false); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> document_reference::get_float32() noexcept {
+  float result;
+  SIMDJSON_TRY(get_float().get(result));
+  return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> document_reference::get_float64() noexcept {
+  double result;
+  SIMDJSON_TRY(get_double().get(result));
+  return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<std::string_view> document_reference::get_string(bool allow_replacement) noexcept { return doc->get_root_value_iterator().get_root_string(false, allow_replacement); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> document_reference::get_u8string(bool allow_replacement) noexcept {
+  std::string_view content;
+  SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+  return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code document_reference::get_string(string_type& receiver, bool allow_replacement) noexcept { return doc->get_root_value_iterator().get_root_string(receiver, false, allow_replacement); }
 simdjson_inline simdjson_result<std::string_view> document_reference::get_wobbly_string() noexcept { return doc->get_root_value_iterator().get_root_wobbly_string(false); }
@@ -97640,11 +123669,25 @@ template<> simdjson_inline simdjson_result<array> document_reference::get() & no
 template<> simdjson_inline simdjson_result<object> document_reference::get() & noexcept { return get_object(); }
 template<> simdjson_inline simdjson_result<raw_json_string> document_reference::get() & noexcept { return get_raw_json_string(); }
 template<> simdjson_inline simdjson_result<std::string_view> document_reference::get() & noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> document_reference::get() & noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_inline simdjson_result<double> document_reference::get() & noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> document_reference::get() & noexcept { return get_float(); }
 template<> simdjson_inline simdjson_result<uint64_t> document_reference::get() & noexcept { return get_uint64(); }
 template<> simdjson_inline simdjson_result<int64_t> document_reference::get() & noexcept { return get_int64(); }
 template<> simdjson_inline simdjson_result<uint32_t> document_reference::get() & noexcept { return get_uint32(); }
 template<> simdjson_inline simdjson_result<int32_t> document_reference::get() & noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> document_reference::get() & noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> document_reference::get() & noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> document_reference::get() & noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> document_reference::get() & noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> document_reference::get() & noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> document_reference::get() & noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<> simdjson_inline simdjson_result<bool> document_reference::get() & noexcept { return get_bool(); }
 template<> simdjson_inline simdjson_result<value> document_reference::get() & noexcept { return get_value(); }
 #if SIMDJSON_EXCEPTIONS
@@ -97790,6 +123833,22 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<haswell::ondemand::docu
   if (error()) { return error(); }
   return first.get_int32();
 }
+simdjson_inline simdjson_result<uint16_t> simdjson_result<haswell::ondemand::document_reference>::get_uint16() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<haswell::ondemand::document_reference>::get_int16() noexcept {
+  if (error()) { return error(); }
+  return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<haswell::ondemand::document_reference>::get_uint8() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<haswell::ondemand::document_reference>::get_int8() noexcept {
+  if (error()) { return error(); }
+  return first.get_int8();
+}
 simdjson_inline simdjson_result<double> simdjson_result<haswell::ondemand::document_reference>::get_double() noexcept {
   if (error()) { return error(); }
   return first.get_double();
@@ -97798,10 +123857,36 @@ simdjson_inline simdjson_result<double> simdjson_result<haswell::ondemand::docum
   if (error()) { return error(); }
   return first.get_double_in_string();
 }
+simdjson_inline simdjson_result<float> simdjson_result<haswell::ondemand::document_reference>::get_float() noexcept {
+  if (error()) { return error(); }
+  return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<haswell::ondemand::document_reference>::get_float_in_string() noexcept {
+  if (error()) { return error(); }
+  return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<haswell::ondemand::document_reference>::get_float32() noexcept {
+  if (error()) { return error(); }
+  return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<haswell::ondemand::document_reference>::get_float64() noexcept {
+  if (error()) { return error(); }
+  return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<std::string_view> simdjson_result<haswell::ondemand::document_reference>::get_string(bool allow_replacement) noexcept {
   if (error()) { return error(); }
   return first.get_string(allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<haswell::ondemand::document_reference>::get_u8string(bool allow_replacement) noexcept {
+  if (error()) { return error(); }
+  return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code simdjson_result<haswell::ondemand::document_reference>::get_string(string_type& receiver, bool allow_replacement) noexcept {
   if (error()) { return error(); }
@@ -97828,22 +123913,46 @@ simdjson_inline simdjson_result<bool> simdjson_result<haswell::ondemand::documen
   return first.is_null();
 }
 template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<haswell::ondemand::document_reference>::get() & noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<haswell::ondemand::document_reference>::get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, haswell::ondemand::document_reference>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>();
 }
 template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<haswell::ondemand::document_reference>::get() && noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<haswell::ondemand::document_reference>::get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, haswell::ondemand::document_reference>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return std::forward<haswell::ondemand::document_reference>(first).get<T>();
 }
 template <class T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<haswell::ondemand::document_reference>::get(T &out) & noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<haswell::ondemand::document_reference>::get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, haswell::ondemand::document_reference>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>(out);
 }
 template <class T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<haswell::ondemand::document_reference>::get(T &out) && noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<haswell::ondemand::document_reference>::get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, haswell::ondemand::document_reference>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return std::forward<haswell::ondemand::document_reference>(first).get<T>(out);
 }
@@ -97905,27 +124014,27 @@ simdjson_inline simdjson_result<haswell::ondemand::document_reference>::operator
 }
 simdjson_inline simdjson_result<haswell::ondemand::document_reference>::operator uint64_t() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_uint64();
 }
 simdjson_inline simdjson_result<haswell::ondemand::document_reference>::operator int64_t() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_int64();
 }
 simdjson_inline simdjson_result<haswell::ondemand::document_reference>::operator double() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_double();
 }
 simdjson_inline simdjson_result<haswell::ondemand::document_reference>::operator std::string_view() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_string();
 }
 simdjson_inline simdjson_result<haswell::ondemand::document_reference>::operator haswell::ondemand::raw_json_string() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_raw_json_string();
 }
 simdjson_inline simdjson_result<haswell::ondemand::document_reference>::operator bool() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_bool();
 }
 simdjson_inline simdjson_result<haswell::ondemand::document_reference>::operator haswell::ondemand::value() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
@@ -97991,6 +124100,7 @@ simdjson_warn_unused simdjson_inline error_code simdjson_result<haswell::ondeman
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */

 #include <algorithm>
+#include <cstring>
 #include <stdexcept>

 namespace simdjson {
@@ -98077,23 +124187,20 @@ simdjson_inline document_stream::document_stream(
   const uint8_t *_buf,
   size_t _len,
   size_t _batch_size,
-  bool _allow_comma_separated
+  bool _allow_comma_separated,
+  stream_format _format
 ) noexcept
   : parser{&_parser},
     buf{_buf},
     len{_len},
     batch_size{_batch_size <= MINIMAL_BATCH_SIZE ? MINIMAL_BATCH_SIZE : _batch_size},
     allow_comma_separated{_allow_comma_separated},
+    format{_format},
     error{SUCCESS}
     #ifdef SIMDJSON_THREADS_ENABLED
     , use_thread(_parser.threaded) // we need to make a copy because _parser.threaded can change
     #endif
 {
-#ifdef SIMDJSON_THREADS_ENABLED
-  if(worker.get() == nullptr) {
-    error = MEMALLOC;
-  }
-#endif
 }

 simdjson_inline document_stream::document_stream() noexcept
@@ -98102,6 +124209,7 @@ simdjson_inline document_stream::document_stream() noexcept
     len{0},
     batch_size{0},
     allow_comma_separated{false},
+    format{stream_format::whitespace_delimited},
     error{UNINITIALIZED}
     #ifdef SIMDJSON_THREADS_ENABLED
     , use_thread(false)
@@ -98121,6 +124229,9 @@ inline size_t document_stream::size_in_bytes() const noexcept {
 }

 inline size_t document_stream::truncated_bytes() const noexcept {
+  // Stage 1 returns EMPTY on zero-length input before it writes the index
+  // sentinels read below, so they would still hold a previous stream's values.
+  if (len == 0) { return 0; }
   if(error == CAPACITY) { return len - batch_start; }
   return parser->implementation->structural_indexes[parser->implementation->n_structural_indexes] - parser->implementation->structural_indexes[parser->implementation->n_structural_indexes + 1];
 }
@@ -98201,13 +124312,20 @@ inline void document_stream::start() noexcept {
     error = run_stage1(*parser, batch_start);
   }
   if (error) { return; }
-  doc_index = batch_start;
+  // For json_sequence mode, structural_indexes[0] points to the actual JSON value
+  // after the RS delimiter and any following whitespace. For regular mode, it is
+  // the offset from batch_start to the first document in the batch.
+  doc_index = batch_start + parser->implementation->structural_indexes[0];
   doc = document(json_iterator(&buf[batch_start], parser));
   doc.iter._streaming = true;

   #ifdef SIMDJSON_THREADS_ENABLED
   if (use_thread && next_batch_start() < len) {
     // Kick off the first thread on next batch if needed
+    if (worker.get() == nullptr) {
+      worker.reset(new(std::nothrow) stage1_worker());
+      if (worker.get() == nullptr) { error = MEMALLOC; return; }
+    }
     error = stage1_thread_parser.allocate(batch_size);
     if (error) { return; }
     worker->start_thread();
@@ -98282,12 +124400,69 @@ inline void document_stream::next() noexcept {
        */

       if (error) { continue; } // If the error was EMPTY, we may want to load another batch.
-      doc_index = batch_start;
+      doc_index = batch_start + parser->implementation->structural_indexes[0];
     }
   }
 }

+simdjson_inline uint8_t document_stream::document_delimiter() const noexcept {
+  switch (format) {
+    case stream_format::newline_delimited: return '\n';
+    case stream_format::json_sequence: return 0x1E;
+    default: return 0;
+  }
+}
+
+simdjson_inline bool document_stream::skip_to_delimiter(uint8_t delimiter) noexcept {
+  const uint8_t *const base = &buf[batch_start];
+  const token_position pos = doc.iter.position();
+  const token_position end = doc.iter.end_position();
+  if (pos >= end) { return false; }
+  const size_t here = size_t(doc.iter.token.peek(pos) - base);
+  const size_t batch_len =
+      (len - batch_start < batch_size) ? len - batch_start : batch_size;
+  if (here >= batch_len) { return false; }
+  const uint8_t *const found = static_cast<const uint8_t *>(
+      std::memchr(base + here, delimiter, batch_len - here));
+  if (found == nullptr) { return false; }
+
+  const uint32_t boundary = uint32_t(found - base);
+  // The answer is near `pos`: the delimiter ends the current document, while
+  // `end` spans the whole batch. Gallop first so the cost follows the distance
+  // rather than the size of the batch.
+  token_position lo = pos;
+  size_t hop = 1;
+  while (lo + hop < end && lo[hop] < boundary) { lo += hop; hop <<= 1; }
+  token_position hi = (lo + hop < end) ? lo + hop : end;
+  while (lo < hi) {
+    const token_position mid = lo + ((hi - lo) >> 1);
+    if (*mid < boundary) { lo = mid + 1; } else { hi = mid; }
+  }
+  doc.iter.token.set_position(lo);
+  return true;
+}
+
 inline void document_stream::next_document() noexcept {
+  // A delimiter that cannot occur inside a document tells us where the current
+  // one ends, so we can jump there instead of walking every structural. Only
+  // valid while the iterator is still inside the document: a consumed document
+  // already sits on the next one's first token, and skip_child() returns at
+  // once for it.
+  //
+  // The jump does not structure-validate the unread remainder of the current
+  // document: under newline_delimited / json_sequence the next delimiter is
+  // assumed to be the true document boundary. Callers that leave depth() > 0
+  // while violating that contract (e.g. pretty multi-line JSON under
+  // newline_delimited) can mis-align following documents; use
+  // whitespace_delimited if unsure.
+  const uint8_t delimiter = document_delimiter();
+  if (delimiter != 0 && !error && doc.iter.depth() > 0 &&
+      skip_to_delimiter(delimiter)) {
+    doc.iter._depth = 1;
+    doc.iter._string_buf_loc = parser->string_buf.get();
+    doc.iter._root = doc.iter.position();
+    return;
+  }
   // Go to next place where depth=0 (document depth)
   error = doc.iter.skip_child(0);
   if (error) { return; }
@@ -98311,10 +124486,35 @@ inline error_code document_stream::run_stage1(ondemand::parser &p, size_t _batch
   // This code only updates the structural index in the parser, it does not update any json_iterator
   // instance.
   size_t remaining = len - _batch_start;
+  stage1_mode mode;
   if (remaining <= batch_size) {
-    return p.implementation->stage1(&buf[_batch_start], remaining, stage1_mode::streaming_final);
+    // Final batch
+    switch (format) {
+      case stream_format::json_sequence:
+        mode = stage1_mode::json_sequence_final;
+        break;
+      case stream_format::comma_delimited:
+        mode = stage1_mode::comma_delimited_final;
+        break;
+      default:
+        mode = stage1_mode::streaming_final;
+        break;
+    }
+    return p.implementation->stage1(&buf[_batch_start], remaining, mode);
   } else {
-    return p.implementation->stage1(&buf[_batch_start], batch_size, stage1_mode::streaming_partial);
+    // Partial batch
+    switch (format) {
+      case stream_format::json_sequence:
+        mode = stage1_mode::json_sequence_partial;
+        break;
+      case stream_format::comma_delimited:
+        mode = stage1_mode::comma_delimited_partial;
+        break;
+      default:
+        mode = stage1_mode::streaming_partial;
+        break;
+    }
+    return p.implementation->stage1(&buf[_batch_start], batch_size, mode);
   }
 }

@@ -98323,11 +124523,19 @@ simdjson_inline size_t document_stream::iterator::current_index() const noexcept
 }

 simdjson_inline std::string_view document_stream::iterator::source() const noexcept {
-  auto depth = stream->doc.iter.depth();
+  // On error (e.g., CAPACITY), there is no document to walk: return the rest of
+  // the input, as the DOM document_stream does.
+  if (stream->error) {
+    return std::string_view(reinterpret_cast<const char*>(stream->buf) + current_index(), stream->len - current_index());
+  }
+  // Always walk from the root of the document, whatever the current position
+  // of the document iterator: the user may have already consumed part of the
+  // document, so the iterator's current depth must not be used here.
+  depth_t depth = 1;
   auto cur_struct_index = stream->doc.iter._root - stream->parser->implementation->structural_indexes.get();

-  // If at root, process the first token to determine if scalar value
-  if (stream->doc.iter.at_root()) {
+  // Process the first token to determine if scalar value
+  {
     switch (stream->buf[stream->batch_start + stream->parser->implementation->structural_indexes[cur_struct_index]]) {
       case '{': case '[':   // Depth=1 already at start of document
         break;
@@ -98335,14 +124543,47 @@ simdjson_inline std::string_view document_stream::iterator::source() const noexc
         depth--;
         break;
       default:    // Scalar value document
-        // TODO: We could remove trailing whitespaces
         // This returns a string spanning from start of value to the beginning of the next document (excluded)
         {
           auto next_index = stream->batch_start + stream->parser->implementation->structural_indexes[++cur_struct_index];
           // normally the length would be next_index - current_index() - 1, except for the last document
           size_t svlen = next_index - current_index();
           const char *start = reinterpret_cast<const char*>(stream->buf) + current_index();
-          while(svlen > 1 && (std::isspace(start[svlen-1]) || start[svlen-1] == '\0')) {
+          // When the scalar is followed by a truncated document, the structural
+          // indexes of that document were dropped and next_index is the end of
+          // the input, so we bound the scalar by scanning the token itself.
+          size_t token_len = 0;
+          if (*start == '"') {
+            token_len = 1;
+            while (token_len < svlen) {
+              char c = start[token_len++];
+              if (c == '\\') {
+                token_len++;
+              } else if (c == '"') {
+                break;
+              }
+            }
+          } else {
+            while (token_len < svlen) {
+              char c = start[token_len];
+              if (std::isspace(static_cast<unsigned char>(c)) || c == ',' || c == '{' || c == '[' || c == '\0' || static_cast<uint8_t>(c) == 0x1E) {
+                break;
+              }
+              token_len++;
+            }
+          }
+          if (token_len > 0 && token_len < svlen) {
+            svlen = token_len;
+          }
+          // Trim trailing whitespace, NUL, and RS (0x1E). In RFC 7464
+          // json_sequence mode the scanner classifies RS as a scalar
+          // character, so an RS-prefixed scalar document (number / true /
+          // false / null / string) has no closing structural index and the
+          // slice runs all the way up to the next document's RS. RS cannot
+          // legally appear in a JSON value at the source level (control
+          // characters in strings must be escaped as \u001E), so stripping
+          // it is safe in every stream_format.
+          while(svlen > 1 && (std::isspace(static_cast<unsigned char>(start[svlen-1])) || start[svlen-1] == '\0' || static_cast<uint8_t>(start[svlen-1]) == 0x1E || (stream->format == stream_format::comma_delimited && start[svlen-1] == ','))) {
             svlen--;
           }
           return std::string_view(start, svlen);
@@ -98467,11 +124708,19 @@ simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> field::un
   return answer;
 }

+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_warn_unused simdjson_result<std::u8string_view> field::unescaped_u8key(bool allow_replacement) noexcept {
+  std::string_view key;
+  SIMDJSON_TRY( unescaped_key(allow_replacement).get(key) );
+  return internal::as_u8string_view(key);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 template <typename string_type>
 simdjson_inline simdjson_warn_unused error_code field::unescaped_key(string_type& receiver, bool allow_replacement) noexcept {
   std::string_view key;
   SIMDJSON_TRY( unescaped_key(allow_replacement).get(key) );
-  receiver = key;
+  internal::assign_utf8(receiver, key);
   return SUCCESS;
 }

@@ -98493,6 +124742,12 @@ simdjson_inline std::string_view field::escaped_key() const noexcept {
   return std::string_view(reinterpret_cast<const char*>(first.buf), end_quote - first.buf);
 }

+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline std::u8string_view field::escaped_u8key() const noexcept {
+  return internal::as_u8string_view(escaped_key());
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 simdjson_inline value &field::value() & noexcept {
   return second;
 }
@@ -98537,11 +124792,25 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<haswell::ondem
   return first.escaped_key();
 }

+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<haswell::ondemand::field>::escaped_u8key() noexcept {
+  if (error()) { return error(); }
+  return first.escaped_u8key();
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 simdjson_inline simdjson_result<std::string_view> simdjson_result<haswell::ondemand::field>::unescaped_key(bool allow_replacement) noexcept {
   if (error()) { return error(); }
   return first.unescaped_key(allow_replacement);
 }

+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<haswell::ondemand::field>::unescaped_u8key(bool allow_replacement) noexcept {
+  if (error()) { return error(); }
+  return first.unescaped_u8key(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 template<typename string_type>
 simdjson_warn_unused simdjson_inline error_code simdjson_result<haswell::ondemand::field>::unescaped_key(string_type &receiver, bool allow_replacement) noexcept {
   if (error()) { return error(); }
@@ -98585,6 +124854,9 @@ simdjson_inline json_iterator::json_iterator(json_iterator &&other) noexcept
     _depth{other._depth},
     _root{other._root},
     _streaming{other._streaming}
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+    , _allow_incomplete_json{other._allow_incomplete_json}
+#endif
 {
   other.parser = nullptr;
 }
@@ -98596,6 +124868,9 @@ simdjson_inline json_iterator &json_iterator::operator=(json_iterator &&other) n
   _depth = other._depth;
   _root = other._root;
   _streaming = other._streaming;
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  _allow_incomplete_json = other._allow_incomplete_json;
+#endif
   other.parser = nullptr;
   return *this;
 }
@@ -98622,7 +124897,8 @@ simdjson_inline json_iterator::json_iterator(const uint8_t *buf, ondemand::parse
       _string_buf_loc{parser->string_buf.get()},
       _depth{1},
       _root{parser->implementation->structural_indexes.get()},
-      _streaming{streaming}
+      _streaming{streaming},
+      _allow_incomplete_json{true}

 {
   logger::log_headers();
@@ -98694,7 +124970,8 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::skip_child(depth_
 #endif // SIMDJSON_CHECK_EOF
       break;
     case '"':
-      if(*peek() == ':') {
+      // At the end, peek() would read the sentinel, which points into the padding.
+      if(!at_end() && *peek() == ':') {
         // We are at a key!!!
         // This might happen if you just started an object and you skip it immediately.
         // Performance note: it would be nice to get rid of this check as it is somewhat
@@ -98737,7 +125014,7 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::skip_child(depth_
     }
   }

-  return report_error(TAPE_ERROR, "not enough close braces");
+  return report_error(INCOMPLETE_ARRAY_OR_OBJECT, "not enough close braces");
 }

 SIMDJSON_POP_DISABLE_WARNINGS
@@ -98754,6 +125031,17 @@ simdjson_inline bool json_iterator::streaming() const noexcept {
   return _streaming;
 }

+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+simdjson_inline bool json_iterator::allow_incomplete_json() const noexcept {
+  return _allow_incomplete_json;
+}
+
+simdjson_inline size_t json_iterator::remaining_input_length(const uint8_t *json) const noexcept {
+  const uint8_t *end = token.buf + parser->_document_len;
+  return json < end ? size_t(end - json) : 0;
+}
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+
 simdjson_inline token_position json_iterator::root_position() const noexcept {
   return _root;
 }
@@ -99036,7 +125324,7 @@ inline std::ostream& operator<<(std::ostream& out, json_type type) noexcept {
         case json_type::string: out << "string"; break;
         case json_type::boolean: out << "boolean"; break;
         case json_type::null: out << "null"; break;
-        default: SIMDJSON_UNREACHABLE();
+        case json_type::unknown: out << "unknown"; break;
     }
     return out;
 }
@@ -99375,6 +125663,10 @@ inline void log_line(const json_iterator &iter, token_position index, depth_t de
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_iterator.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value-inl.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/jsonpathutil.h" */
+/* amalgamation skipped (editor-only): #include <utility> */
+/* amalgamation skipped (editor-only): #if SIMDJSON_SUPPORTS_CONCEPTS */
+/* amalgamation skipped (editor-only): #include <tuple> // std::forward_as_tuple/get for the variadic for_each adapter */
+/* amalgamation skipped (editor-only): #endif */
 /* amalgamation skipped (editor-only): #if SIMDJSON_STATIC_REFLECTION */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_string_builder.h"  // for constevalutil::fixed_string */
 /* amalgamation skipped (editor-only): #include <meta> */
@@ -99404,12 +125696,21 @@ simdjson_inline simdjson_result<value> object::find_field_unordered(const std::s
   return value(iter.child());
 }
 simdjson_inline simdjson_result<value> object::operator[](const std::string_view key) & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   return find_field_unordered(key);
 }
 simdjson_inline simdjson_result<value> object::operator[](const std::string_view key) && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   return std::forward<object>(*this).find_field_unordered(key);
 }
 simdjson_inline simdjson_result<value> object::find_field(const std::string_view key) & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   bool has_value;
   SIMDJSON_TRY( iter.find_field_raw(key).get(has_value) );
   if (!has_value) {
@@ -99419,6 +125720,9 @@ simdjson_inline simdjson_result<value> object::find_field(const std::string_view
   return value(iter.child());
 }
 simdjson_inline simdjson_result<value> object::find_field(const std::string_view key) && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   bool has_value;
   SIMDJSON_TRY( iter.find_field_raw(key).get(has_value) );
   if (!has_value) {
@@ -99428,6 +125732,150 @@ simdjson_inline simdjson_result<value> object::find_field(const std::string_view
   return value(iter.child());
 }

+#if SIMDJSON_SUPPORTS_CONCEPTS
+template <typename Selector, typename Func>
+  requires key_selector_type<Selector> &&
+           std::is_invocable_v<Func&, std::size_t, value>
+simdjson_flatten simdjson_inline for_each_result object::for_each(Func&& on_match)
+    noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, value>) {
+  // Single pass driven directly by the value_iterator, mirroring
+  // find_field_unordered_raw + value(iter.child()). Compared to walking via
+  // object_iterator/field, this avoids constructing a simdjson_result<field> and
+  // a field (key + value) for every field -- and the development-check bookkeeping
+  // in object_iterator -- building a value only for the fields that actually match.
+  // We operate on a copy of the iterator, as object::begin() would.
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  // Mirror object::begin(): for_each must start at the beginning of the object,
+  // not from some position left behind by a prior find_field on the same object.
+  if (!iter.is_at_iterator_start()) { return {OUT_OF_ORDER_ITERATION, 0}; }
+#endif
+  value_iterator it = iter;
+  std::size_t matched = 0;
+  // Track which selector indices have already matched, as a compile-time bitset
+  // (one 64-bit word per 64 keys): the callback fires once per key, on its first
+  // occurrence, and we stop as soon as every key has matched.
+  constexpr std::size_t seen_words = (Selector::size() + 63) / 64;
+  std::array<std::uint64_t, seen_words> seen{};
+  while (it.is_open()) {
+    raw_json_string key;
+    error_code error;
+    std::size_t idx;
+    if constexpr (Selector::window.ok) {
+      // A window selector confirms a key from its raw bytes alone (the closing
+      // quote bounds it), so we take the length-free path: field_key (no backward
+      // scan, mirroring the ordered find_field) feeding match_raw(raw_json_string).
+      if ((error = it.field_key().get(key))) { it.abandon(); return {error, matched}; }
+      if ((error = it.field_value())) { it.abandon(); return {error, matched}; }
+      idx = Selector::match_raw(key);
+    } else {
+      // Otherwise derive the key length from the structural index (the following
+      // ':' token) to feed the length-aware hash, avoiding a forward SIMD scan.
+      std::size_t key_len;
+      if ((error = it.field_key_with_length(key, key_len))) { it.abandon(); return {error, matched}; }
+      if ((error = it.field_value())) { it.abandon(); return {error, matched}; }
+      idx = Selector::match_raw(key.raw(), key_len);
+    }
+    if (idx < Selector::size()) {
+      const std::uint64_t seen_bit = std::uint64_t{1} << (idx & 63);
+      std::uint64_t &seen_word = seen[idx >> 6];
+      if (!(seen_word & seen_bit)) {
+        seen_word |= seen_bit;
+        value matched_value(it.child());
+        // The callback may return void or anything convertible to error_code
+        // (error_code itself, or a for_each_result from a nested for_each). When
+        // it yields an error_code, we stop at the first non-SUCCESS result and
+        // propagate it so the caller can surface value-parse errors (for example,
+        // a type mismatch on a matched field). A void-returning callback is
+        // responsible for handling its own errors.
+        if constexpr (std::is_convertible_v<decltype(on_match(idx, matched_value)), error_code>) {
+          // Unlike the internal-error paths above, a callback error does not
+          // abandon the iterator: we leave it recoverable so the caller can keep
+          // using the object (or its parent) after handling the error.
+          if ((error = on_match(idx, matched_value))) { return {error, matched}; }
+        } else {
+          on_match(idx, matched_value);
+        }
+        if (++matched >= Selector::size()) { break; }
+      }
+    }
+    // Mirror object_iterator::operator++'s safety rail: if the callback consumed
+    // the value and left the iterator closed or in error (e.g. a void callback
+    // that swallowed a fatal sub-iteration error), stop here rather than calling
+    // skip_child on a closed iterator.
+    if (!it.is_open()) { break; }
+    // Skip the value (a no-op if the callback consumed it) and step to the next
+    // field; has_next_field() ends the container on '}', which closes the loop.
+    if ((error = it.skip_child())) { it.abandon(); return {error, matched}; }
+    if ((error = it.has_next_field().error())) { return {error, matched}; }
+  }
+  return {SUCCESS, matched};
+}
+
+namespace key_selector_for_each_detail {
+
+// Dispatch a matched value to the I-th handler in the tuple (0-based). A handler
+// is either an invocable taking the value (void- or error_code-returning) or a
+// deserialization target, in which case we assign via value::get. We always
+// return error_code so the core (index, value) for_each can treat the adapter
+// uniformly.
+template <typename Tuple, std::size_t... Is>
+simdjson_really_inline error_code dispatch_value(
+    std::size_t idx, Tuple& handlers, value v, std::index_sequence<Is...>) {
+  error_code err = SUCCESS;
+  auto try_one = [&](auto Ic) {
+    constexpr std::size_t I = decltype(Ic)::value;
+    if (idx == I) {
+      auto&& h = std::get<I>(handlers);
+      using H = std::remove_reference_t<decltype(h)>;
+      if constexpr (std::is_invocable_v<H&, value>) {
+        // A handler returning void runs for its side effects; one returning
+        // anything convertible to error_code (error_code, or a for_each_result
+        // from a nested for_each) has its error captured and propagated.
+        if constexpr (std::is_convertible_v<decltype(h(v)), error_code>) {
+          err = h(v);
+        } else {
+          h(v);
+        }
+      } else {
+        // Direct deserialization target: assign the matched value into it.
+        err = v.get(h);
+      }
+    }
+  };
+  (try_one(std::integral_constant<std::size_t, Is>{}), ...);
+  return err;
+}
+
+} // namespace key_selector_for_each_detail
+
+template <typename Selector, typename... Handlers>
+  requires key_selector_type<Selector> &&
+           (sizeof...(Handlers) == Selector::size()) &&
+           (key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_flatten simdjson_inline for_each_result object::for_each(Handlers&&... on_match)
+    noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+  auto handlers = std::forward_as_tuple(std::forward<Handlers>(on_match)...);
+  // Reuse the single (index, value) implementation via a tiny adapter.
+  // The adapter is called once per *matched* key (very few); the hot path
+  // (iteration + match_raw + seen bitset) stays exactly the same.
+  return this->template for_each<Selector>(
+      [&](std::size_t i, value v) -> error_code {
+        return key_selector_for_each_detail::dispatch_value(
+            i, handlers, v, std::make_index_sequence<Selector::size()>{});
+      });
+}
+
+template <constevalutil::fixed_string... Keys, typename... Handlers>
+  requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+           (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+           (key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline for_each_result object::for_each(Handlers&&... on_match)
+    noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+  using Selector = key_selector<Keys...>;
+  return this->template for_each<Selector>(std::forward<Handlers>(on_match)...);
+}
+#endif
+
 simdjson_inline simdjson_result<object> object::start(value_iterator &iter) noexcept {
   SIMDJSON_TRY( iter.start_object().error() );
   return object(iter);
@@ -99463,6 +125911,9 @@ simdjson_warn_unused simdjson_inline error_code object::consume() noexcept {
 }

 simdjson_inline simdjson_result<std::string_view> object::raw_json() noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   const uint8_t * starting_point{iter.peek_start()};
   auto error = consume();
   if(error) { return error; }
@@ -99484,9 +125935,17 @@ simdjson_inline object::object(const value_iterator &_iter) noexcept
 {
 }

-simdjson_inline simdjson_result<object_iterator> object::begin() noexcept {
+simdjson_inline simdjson_result<object_iterator> object::begin() & noexcept {
 #if SIMDJSON_DEVELOPMENT_CHECKS
-  if (!iter.is_at_iterator_start()) { return OUT_OF_ORDER_ITERATION; }
+  if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
+  return object_iterator(iter, this);
+#endif
+  return object_iterator(iter);
+}
+simdjson_inline simdjson_result<object_iterator> object::begin() && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  // The object is a temporary that the iterator may outlive: do not lock it.
+  if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
 #endif
   return object_iterator(iter);
 }
@@ -99495,7 +125954,9 @@ simdjson_inline simdjson_result<object_iterator> object::end() noexcept {
 }

 inline simdjson_result<value> object::at_pointer(std::string_view json_pointer) noexcept {
-  if (json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
+  // An empty pointer has no json_pointer[0]: with a default-constructed
+  // std::string_view, reading it dereferences a null pointer.
+  if (json_pointer.empty() || json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
   json_pointer = json_pointer.substr(1);
   size_t slash = json_pointer.find('/');
   std::string_view key = json_pointer.substr(0, slash);
@@ -99597,6 +126058,9 @@ simdjson_inline simdjson_result<size_t> object::count_fields() & noexcept {
 }

 simdjson_inline simdjson_result<bool> object::is_empty() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   bool is_not_empty;
   auto error = iter.reset_object().get(is_not_empty);
   if(error) { return error; }
@@ -99604,9 +126068,41 @@ simdjson_inline simdjson_result<bool> object::is_empty() & noexcept {
 }

 simdjson_inline simdjson_result<bool> object::reset() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   return iter.reset_object();
 }

+simdjson_inline object_position object::get_current_position() const noexcept {
+  return object_position{iter.position(), iter.json_iter().depth()};
+}
+
+simdjson_inline error_code object::revert_position(object_position position) noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
+  // json_iterator::reenter_child() requires the live depth to be exactly
+  // one level shallower than the target (matching how every other depth
+  // transition in this iterator works), and under SIMDJSON_DEVELOPMENT_CHECKS
+  // additionally validates against the parser's per-depth container-start
+  // bookkeeping. Neither applies here: depending on what was captured and
+  // what has happened since (a scalar field fully consumed, a compound
+  // value left open, a find_field() miss that scanned past everything),
+  // the live depth when reverting can be any number of levels away from
+  // the captured one, and the captured depth is not necessarily a
+  // container's own start. reenter_at() moves directly, matching how
+  // reset_object() itself repositions without going through reenter_child().
+  iter.reenter_at(position.position, position.depth);
+  return SUCCESS;
+}
+
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline void object::set_locked(bool _locked) noexcept {
+  locked = _locked;
+}
+#endif
+
 #if SIMDJSON_SUPPORTS_CONCEPTS && SIMDJSON_STATIC_REFLECTION

 template<constevalutil::fixed_string... FieldNames, typename T>
@@ -99664,10 +126160,14 @@ simdjson_inline simdjson_result<haswell::ondemand::object>::simdjson_result(hasw
 simdjson_inline simdjson_result<haswell::ondemand::object>::simdjson_result(error_code error) noexcept
     : implementation_simdjson_result_base<haswell::ondemand::object>(error) {}

-simdjson_inline simdjson_result<haswell::ondemand::object_iterator> simdjson_result<haswell::ondemand::object>::begin() noexcept {
+simdjson_inline simdjson_result<haswell::ondemand::object_iterator> simdjson_result<haswell::ondemand::object>::begin() & noexcept {
   if (error()) { return error(); }
   return first.begin();
 }
+simdjson_inline simdjson_result<haswell::ondemand::object_iterator> simdjson_result<haswell::ondemand::object>::begin() && noexcept {
+  if (error()) { return error(); }
+  return std::move(first).begin();
+}
 simdjson_inline simdjson_result<haswell::ondemand::object_iterator> simdjson_result<haswell::ondemand::object>::end() noexcept {
   if (error()) { return error(); }
   return first.end();
@@ -99721,11 +126221,55 @@ simdjson_inline error_code simdjson_result<haswell::ondemand::object>::for_each_
   return first.for_each_at_path_with_wildcard(json_path, std::forward<Func>(callback));
 }

+#if SIMDJSON_SUPPORTS_CONCEPTS
+template <typename Selector, typename Func>
+  requires haswell::ondemand::key_selector_type<Selector> &&
+           std::is_invocable_v<Func&, std::size_t, haswell::ondemand::value>
+simdjson_inline haswell::ondemand::for_each_result
+simdjson_result<haswell::ondemand::object>::for_each(Func&& on_match)
+    noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, haswell::ondemand::value>) {
+  if (error()) { return {error(), 0}; }
+  return first.template for_each<Selector>(std::forward<Func>(on_match));
+}
+
+template <typename Selector, typename... Handlers>
+  requires haswell::ondemand::key_selector_type<Selector> &&
+           (sizeof...(Handlers) == Selector::size()) &&
+           (haswell::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline haswell::ondemand::for_each_result
+simdjson_result<haswell::ondemand::object>::for_each(Handlers&&... on_match)
+    noexcept(haswell::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+  if (error()) { return {error(), 0}; }
+  return first.template for_each<Selector>(std::forward<Handlers>(on_match)...);
+}
+
+template <constevalutil::fixed_string... Keys, typename... Handlers>
+  requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+           (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+           (haswell::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline haswell::ondemand::for_each_result
+simdjson_result<haswell::ondemand::object>::for_each(Handlers&&... on_match)
+    noexcept(haswell::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+  if (error()) { return {error(), 0}; }
+  return first.template for_each<Keys...>(std::forward<Handlers>(on_match)...);
+}
+#endif
+
 inline simdjson_result<bool> simdjson_result<haswell::ondemand::object>::reset() noexcept {
   if (error()) { return error(); }
   return first.reset();
 }

+inline simdjson_result<haswell::ondemand::object_position> simdjson_result<haswell::ondemand::object>::get_current_position() noexcept {
+  if (error()) { return error(); }
+  return first.get_current_position();
+}
+
+inline error_code simdjson_result<haswell::ondemand::object>::revert_position(haswell::ondemand::object_position position) noexcept {
+  if (error()) { return error(); }
+  return first.revert_position(position);
+}
+
 inline simdjson_result<bool> simdjson_result<haswell::ondemand::object>::is_empty() noexcept {
   if (error()) { return error(); }
   return first.is_empty();
@@ -99769,6 +126313,61 @@ simdjson_inline object_iterator::object_iterator(const value_iterator &_iter) no
   : iter{_iter}
 {}

+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline object_iterator::object_iterator(const value_iterator &_iter, object* _parent) noexcept
+  : parent{_parent}, iter{_iter}
+{
+  if (parent) parent->set_locked(true);
+}
+#endif
+
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline object_iterator::~object_iterator() noexcept
+{
+  if (parent) parent->set_locked(false);
+}
+
+simdjson_inline object_iterator::object_iterator(object_iterator&& other) noexcept
+  : has_been_referenced{other.has_been_referenced},
+    parent{other.parent},
+    iter{std::move(other.iter)}
+{
+  other.parent = nullptr;
+}
+
+simdjson_inline object_iterator& object_iterator::operator=(object_iterator&& other) noexcept {
+  if (this != &other)
+  {
+    if (parent)
+      parent->set_locked(false);
+    has_been_referenced = other.has_been_referenced;
+    parent = other.parent;
+    iter = std::move(other.iter);
+
+    other.parent = nullptr;
+  }
+  return *this;
+}
+
+simdjson_inline object_iterator::object_iterator(const object_iterator& other) noexcept
+  : has_been_referenced{other.has_been_referenced},
+    parent{nullptr},
+    iter{other.iter}
+{}
+
+simdjson_inline object_iterator& object_iterator::operator=(const object_iterator& other) noexcept {
+  if (this != &other)
+  {
+    if (parent)
+      parent->set_locked(false);
+    has_been_referenced = other.has_been_referenced;
+    parent = nullptr;
+    iter = other.iter;
+  }
+  return *this;
+}
+#endif
+
 simdjson_inline simdjson_result<field> object_iterator::operator*() noexcept {
 #if SIMDJSON_DEVELOPMENT_CHECKS
   // We must call * once per iteration.
@@ -99896,6 +126495,147 @@ simdjson_inline simdjson_result<haswell::ondemand::object_iterator> &simdjson_re

 #endif // SIMDJSON_GENERIC_ONDEMAND_OBJECT_ITERATOR_INL_H
 /* end file simdjson/generic/ondemand/object_iterator-inl.h for haswell */
+/* including simdjson/generic/ondemand/ranges-inl.h for haswell: #include "simdjson/generic/ondemand/ranges-inl.h" */
+/* begin file simdjson/generic/ondemand/ranges-inl.h for haswell */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/ranges.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array_iterator-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object_iterator-inl.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#if SIMDJSON_SUPPORTS_RANGES
+
+namespace simdjson {
+namespace haswell {
+namespace ondemand {
+
+//
+// array_range_iterator
+//
+
+simdjson_inline array_range_iterator::array_range_iterator(array_iterator iter) noexcept
+  : iter_{iter} {}
+
+simdjson_inline simdjson_result<value> array_range_iterator::operator*() const noexcept {
+  return *iter_;
+}
+
+simdjson_inline array_range_iterator& array_range_iterator::operator++() noexcept {
+  ++iter_;
+  return *this;
+}
+
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+simdjson_inline void array_range_iterator::operator++(int) noexcept {
+  ++*this;
+}
+SIMDJSON_POP_DISABLE_WARNINGS
+
+//
+// array_range
+//
+
+simdjson_inline array_range::array_range(array& arr) noexcept {
+  auto b = arr.begin();
+  if (b.error()) { error_ = b.error(); return; }
+  begin_ = b.value_unsafe();
+  end_ = arr.end().value_unsafe();
+}
+
+simdjson_inline array_range_iterator array_range::begin() noexcept {
+  return array_range_iterator(begin_);
+}
+
+simdjson_inline array_range_iterator array_range::end() noexcept {
+  return array_range_iterator(end_);
+}
+
+//
+// object_range_iterator
+//
+
+simdjson_inline object_range_iterator::object_range_iterator(object_iterator iter) noexcept
+  : iter_{iter} {}
+
+simdjson_inline simdjson_result<field> object_range_iterator::operator*() const noexcept {
+  return *iter_;
+}
+
+simdjson_inline object_range_iterator& object_range_iterator::operator++() noexcept {
+  ++iter_;
+  return *this;
+}
+
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+simdjson_inline void object_range_iterator::operator++(int) noexcept {
+  ++*this;
+}
+SIMDJSON_POP_DISABLE_WARNINGS
+
+
+//
+// object_range
+//
+
+simdjson_inline object_range::object_range(object& obj) noexcept {
+  auto b = obj.begin();
+  if (b.error()) { error_ = b.error(); return; }
+  begin_ = b.value_unsafe();
+  end_ = obj.end().value_unsafe();
+}
+
+simdjson_inline object_range_iterator object_range::begin() noexcept {
+  return object_range_iterator(begin_);
+}
+
+simdjson_inline object_range_iterator object_range::end() noexcept {
+  return object_range_iterator(end_);
+}
+
+//
+// Free functions
+//
+
+simdjson_inline array_range get_range(array& arr) noexcept {
+  return array_range(arr);
+}
+
+simdjson_inline object_range get_key_value_range(object& obj) noexcept {
+  return object_range(obj);
+}
+
+#if SIMDJSON_EXCEPTIONS
+simdjson_inline array_range get_range(simdjson_result<array> result) {
+  return array_range(result.value());
+}
+
+simdjson_inline object_range get_key_value_range(simdjson_result<object> result) {
+  return object_range(result.value());
+}
+#endif // SIMDJSON_EXCEPTIONS
+
+} // namespace ondemand
+} // namespace haswell
+} // namespace simdjson
+
+// Verify the range wrapper types satisfy the expected C++20 concepts.
+static_assert(std::input_iterator<simdjson::haswell::ondemand::array_range_iterator>);
+static_assert(std::input_iterator<simdjson::haswell::ondemand::object_range_iterator>);
+static_assert(std::ranges::input_range<simdjson::haswell::ondemand::array_range>);
+static_assert(std::ranges::input_range<simdjson::haswell::ondemand::object_range>);
+static_assert(std::ranges::view<simdjson::haswell::ondemand::array_range>);
+static_assert(std::ranges::view<simdjson::haswell::ondemand::object_range>);
+
+#endif // SIMDJSON_SUPPORTS_RANGES
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H
+/* end file simdjson/generic/ondemand/ranges-inl.h for haswell */
 /* including simdjson/generic/ondemand/parser-inl.h for haswell: #include "simdjson/generic/ondemand/parser-inl.h" */
 /* begin file simdjson/generic/ondemand/parser-inl.h for haswell */
 #ifndef SIMDJSON_GENERIC_ONDEMAND_PARSER_INL_H
@@ -99927,7 +126667,10 @@ simdjson_warn_unused simdjson_inline error_code parser::allocate(size_t new_capa

   // string_capacity copied from document::allocate
   _capacity = 0;
-  size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * new_capacity / 3 + SIMDJSON_PADDING, 64);
+  if(5 * (new_capacity / 3) + SIMDJSON_PADDING < SIMDJSON_PADDING) {
+    return CAPACITY; // overflow, only happen on legacy 32-bit systems with very large capacity
+  }
+  size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * (new_capacity / 3) + SIMDJSON_PADDING, 64);
   string_buf.reset(new (std::nothrow) uint8_t[string_capacity]);
 #if SIMDJSON_DEVELOPMENT_CHECKS
   start_positions.reset(new (std::nothrow) token_position[new_max_depth]);
@@ -99952,6 +126695,7 @@ simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate(p
   if (!json.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }

   json.remove_utf8_bom();
+  _document_len = json.length();

   // Allocate if needed
   if (capacity() < json.length() || !string_buf) {
@@ -99968,6 +126712,7 @@ simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate_a
   if (!json.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }

   json.remove_utf8_bom();
+  _document_len = json.length();

   // Allocate if needed
   if (capacity() < json.length() || !string_buf) {
@@ -100033,6 +126778,34 @@ simdjson_warn_unused simdjson_inline simdjson_result<json_iterator> parser::iter
   return json_iterator(reinterpret_cast<const uint8_t *>(json.data()), this);
 }

+inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size) noexcept {
+  if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
+  if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+    buf += 3;
+    len -= 3;
+  }
+  return document_stream(*this, buf, len, batch_size, false, stream_format::whitespace_delimited);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size) noexcept {
+  return iterate_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(padded_string_view s, size_t batch_size) noexcept {
+  if (!s.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
+  return iterate_many(s.data(), s.length(), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const padded_string &s, size_t batch_size) noexcept {
+  return iterate_many(padded_string_view(s), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const std::string &s, size_t batch_size) noexcept {
+  return iterate_many(padded_string_view(s), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size) noexcept {
+  return iterate_many(pad(s), batch_size);
+}
+
+#ifndef SIMDJSON_DISABLE_DEPRECATED_API
+SIMDJSON_PUSH_DISABLE_WARNINGS
+SIMDJSON_DISABLE_DEPRECATED_WARNING
 inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept {
   // Warning: no check is done on the buffer padding. We trust the user.
   if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
@@ -100040,8 +126813,11 @@ inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf,
     buf += 3;
     len -= 3;
   }
-  if(allow_comma_separated && batch_size < len) { batch_size = len; }
-  return document_stream(*this, buf, len, batch_size, allow_comma_separated);
+  // Map allow_comma_separated to stream_format::comma_delimited
+  if (allow_comma_separated) {
+    return document_stream(*this, buf, len, batch_size, false, stream_format::comma_delimited);
+  }
+  return document_stream(*this, buf, len, batch_size, false, stream_format::whitespace_delimited);
 }

 inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept {
@@ -100061,6 +126837,51 @@ inline simdjson_result<document_stream> parser::iterate_many(const std::string &
 inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size, bool allow_comma_separated) noexcept {
   return iterate_many(pad(s), batch_size, allow_comma_separated);
 }
+SIMDJSON_POP_DISABLE_WARNINGS
+#endif // SIMDJSON_DISABLE_DEPRECATED_API
+
+inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+  if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
+  if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+    buf += 3;
+    len -= 3;
+  }
+  if (format == stream_format::comma_delimited_array) {
+    // Strip leading JSON whitespace.
+    while (len > 0 && (buf[0] == ' ' || buf[0] == '\t' || buf[0] == '\n' || buf[0] == '\r')) {
+      buf++; len--;
+    }
+    // Expect the opening '['.
+    if (len == 0 || buf[0] != '[') { return TAPE_ERROR; }
+    buf++; len--;
+    // Strip trailing JSON whitespace.
+    while (len > 0 && (buf[len-1] == ' ' || buf[len-1] == '\t' || buf[len-1] == '\n' || buf[len-1] == '\r')) {
+      len--;
+    }
+    // Expect the closing ']'.
+    if (len == 0 || buf[len-1] != ']') { return TAPE_ERROR; }
+    len--;
+    // Fall through to comma_delimited over the array contents.
+    format = stream_format::comma_delimited;
+  }
+  return document_stream(*this, buf, len, batch_size, false, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+  return iterate_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(padded_string_view s, size_t batch_size, stream_format format) noexcept {
+  if (!s.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
+  return iterate_many(s.data(), s.length(), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const std::string &s, size_t batch_size, stream_format format) noexcept {
+  return iterate_many(padded_string_view(s), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size, stream_format format) noexcept {
+  return iterate_many(pad(s), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const padded_string &s, size_t batch_size, stream_format format) noexcept {
+  return iterate_many(padded_string_view(s), batch_size, format);
+}
 simdjson_pure simdjson_inline size_t parser::capacity() const noexcept {
   return _capacity;
 }
@@ -100468,6 +127289,27 @@ namespace simdjson {
 namespace haswell {
 namespace ondemand {

+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+simdjson_inline bool raw_json_string_is_quote_terminated(const uint8_t *json, uint32_t max_len) noexcept {
+  bool escaping{false};
+  for (uint32_t i = 1; i < max_len; i++) {
+    switch (json[i]) {
+      case '"':
+        if (!escaping) { return true; }
+        escaping = false;
+        break;
+      case '\\':
+        escaping = !escaping;
+        break;
+      default:
+        escaping = false;
+        break;
+    }
+  }
+  return false;
+}
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+
 simdjson_inline value_iterator::value_iterator(
   json_iterator *json_iter,
   depth_t depth,
@@ -100855,6 +127697,22 @@ simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iter
   return raw_json_string(key);
 }

+simdjson_warn_unused simdjson_inline error_code value_iterator::field_key_with_length(raw_json_string &key, std::size_t &len) noexcept {
+  assert_at_next();
+
+  const uint8_t *k = _json_iter->return_current_and_advance();
+  if (*(k++) != '"') { return report_error(TAPE_ERROR, "Object key is not a string"); }
+  // After return_current_and_advance(), the current token is the ':' that follows
+  // the key. The closing quote sits just before it (only JSON whitespace may
+  // intervene), so step back from the ':' to the closing quote to get the length.
+  // In minified JSON this is a single back-step.
+  const char *q = reinterpret_cast<const char *>(_json_iter->peek());
+  do { --q; } while (*q != '"');
+  key = raw_json_string(k);
+  len = static_cast<std::size_t>(q - reinterpret_cast<const char *>(k));
+  return SUCCESS;
+}
+
 simdjson_warn_unused simdjson_inline error_code value_iterator::field_value() noexcept {
   assert_at_next();

@@ -100972,7 +127830,7 @@ simdjson_warn_unused simdjson_inline error_code value_iterator::get_string(strin
   auto saved_string_buf_loc = _json_iter->string_buf_loc();
   auto err = get_string(allow_replacement).get(content);
   if (err) { return err; }
-  receiver = content;
+  internal::assign_utf8(receiver, content);
   // Restore the string buffer location, effectively discarding any temporary string storage
   _json_iter->string_buf_loc() = saved_string_buf_loc;
   return SUCCESS;
@@ -100983,6 +127841,15 @@ simdjson_warn_unused simdjson_inline simdjson_result<std::string_view> value_ite
 simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iterator::get_raw_json_string() noexcept {
   auto json = peek_scalar("string");
   if (*json != '"') { return incorrect_type_error("Not a string"); }
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  if (_json_iter->allow_incomplete_json()) {
+    const size_t remaining_input_length = _json_iter->remaining_input_length(json);
+    const uint32_t max_len = remaining_input_length < peek_start_length() ? uint32_t(remaining_input_length) : peek_start_length();
+    if (!raw_json_string_is_quote_terminated(json, max_len)) {
+      return STRING_ERROR;
+    }
+  }
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
   advance_scalar("string");
   return raw_json_string(json+1);
 }
@@ -101016,6 +127883,16 @@ simdjson_warn_unused simdjson_inline simdjson_result<double> value_iterator::get
   if(result.error() == SUCCESS) { advance_non_root_scalar("double"); }
   return result;
 }
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_float() noexcept {
+  auto result = numberparsing::parse_float(peek_non_root_scalar("float"));
+  if(result.error() == SUCCESS) { advance_non_root_scalar("float"); }
+  return result;
+}
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_float_in_string() noexcept {
+  auto result = numberparsing::parse_float_in_string(peek_non_root_scalar("float"));
+  if(result.error() == SUCCESS) { advance_non_root_scalar("float"); }
+  return result;
+}
 simdjson_warn_unused simdjson_inline simdjson_result<bool> value_iterator::get_bool() noexcept {
   auto result = parse_bool(peek_non_root_scalar("bool"));
   if(result.error() == SUCCESS) { advance_non_root_scalar("bool"); }
@@ -101118,7 +127995,7 @@ simdjson_warn_unused simdjson_inline error_code value_iterator::get_root_string(
   auto saved_string_buf_loc = _json_iter->string_buf_loc();
   auto err = get_root_string(check_trailing, allow_replacement).get(content);
   if (err) { return err; }
-  receiver = content;
+  internal::assign_utf8(receiver, content);
   // Restore the string buffer location, effectively discarding any temporary string storage
   _json_iter->string_buf_loc() = saved_string_buf_loc;
   return SUCCESS;
@@ -101130,6 +128007,15 @@ simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iter
   auto json = peek_scalar("string");
   if (*json != '"') { return incorrect_type_error("Not a string"); }
   if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  if (_json_iter->allow_incomplete_json()) {
+    const size_t remaining_input_length = _json_iter->remaining_input_length(json);
+    const uint32_t max_len = remaining_input_length < peek_root_length() ? uint32_t(remaining_input_length) : peek_root_length();
+    if (!raw_json_string_is_quote_terminated(json, max_len)) {
+      return STRING_ERROR;
+    }
+  }
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
   advance_scalar("string");
   return raw_json_string(json+1);
 }
@@ -101239,6 +128125,43 @@ simdjson_warn_unused simdjson_inline simdjson_result<double> value_iterator::get
   return result;
 }

+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_root_float(bool check_trailing) noexcept {
+  auto max_len = peek_root_length();
+  auto json = peek_root_scalar("float");
+  // We use the same buffer size as get_root_double: the number of significant
+  // digits that matter is smaller for binary32, but the JSON document may still
+  // spell out a long number that we must parse (and round) faithfully.
+  uint8_t tmpbuf[1074+8+1+1]; // +1 for null termination.
+  tmpbuf[1074+8+1] = '\0'; // make sure that buffer is always null terminated.
+  if (!_json_iter->copy_to_buffer(json, max_len, tmpbuf, 1074+8+1)) {
+    logger::log_error(*_json_iter, start_position(), depth(), "Root number more than 1082 characters");
+    return NUMBER_ERROR;
+  }
+  auto result = numberparsing::parse_float(tmpbuf);
+  if(result.error() == SUCCESS) {
+    if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+    advance_root_scalar("float");
+  }
+  return result;
+}
+
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_root_float_in_string(bool check_trailing) noexcept {
+  auto max_len = peek_root_length();
+  auto json = peek_root_scalar("float");
+  uint8_t tmpbuf[1074+8+1+1]; // +1 for null termination.
+  tmpbuf[1074+8+1] = '\0'; // make sure that buffer is always null terminated.
+  if (!_json_iter->copy_to_buffer(json, max_len, tmpbuf, 1074+8+1)) {
+    logger::log_error(*_json_iter, start_position(), depth(), "Root number more than 1082 characters");
+    return NUMBER_ERROR;
+  }
+  auto result = numberparsing::parse_float_in_string(tmpbuf);
+  if(result.error() == SUCCESS) {
+    if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+    advance_root_scalar("float");
+  }
+  return result;
+}
+
 simdjson_warn_unused simdjson_inline simdjson_result<bool> value_iterator::get_root_bool(bool check_trailing) noexcept {
   auto max_len = peek_root_length();
   auto json = peek_root_scalar("bool");
@@ -101467,6 +128390,21 @@ simdjson_inline void value_iterator::move_at_container_start() noexcept {
   _json_iter->token.set_position(_start_position + 1);
 }

+simdjson_inline void value_iterator::reenter_at(token_position position, depth_t depth) noexcept {
+  // Unlike reenter_child(), this does not require the live depth to be
+  // exactly one level shallower than depth, nor does it validate against
+  // the parser's per-depth container-start bookkeeping: neither holds in
+  // general for a caller-supplied snapshot (see object_position). What
+  // must still always hold, regardless of what was captured or how far
+  // the live iterator has since moved, is that position and depth are
+  // themselves sane values -- this is the same bound reenter_child()
+  // itself applies unconditionally.
+  SIMDJSON_ASSUME(position != nullptr);
+  SIMDJSON_ASSUME(depth >= 1 && depth < INT32_MAX);
+  _json_iter->_depth = depth;
+  _json_iter->token.set_position(position);
+}
+
 simdjson_inline simdjson_result<bool> value_iterator::reset_array() noexcept {
   if(error()) { return error(); }
   move_at_container_start();
@@ -102916,16 +129854,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
 }
 #endif

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
-                                uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
-  return _addcarry_u64(0, value1, value2,
-                       reinterpret_cast<unsigned __int64 *>(result));
-#else
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-#endif
-}

 } // unnamed namespace
 } // namespace icelake
@@ -103488,7 +130416,7 @@ simdjson_inline internal::value128 full_multiplication(uint64_t value1, uint64_t
 /* end file simdjson/icelake/begin.h */
 /* including simdjson/generic/ondemand/amalgamated.h for icelake: #include "simdjson/generic/ondemand/amalgamated.h" */
 /* begin file simdjson/generic/ondemand/amalgamated.h for icelake */
-#if defined(SIMDJSON_CONDITIONAL_INCLUDE) && !defined(SIMDJSON_GENERIC_BUILDER_DEPENDENCIES_H)
+#if defined(SIMDJSON_CONDITIONAL_INCLUDE) && !defined(SIMDJSON_GENERIC_ONDEMAND_DEPENDENCIES_H)
 #error simdjson/generic/ondemand/dependencies.h must be included before simdjson/generic/ondemand/amalgamated.h!
 #endif

@@ -103537,6 +130465,13 @@ class token_iterator;
 class value;
 class value_iterator;

+#if SIMDJSON_SUPPORTS_RANGES
+class array_range;
+class array_range_iterator;
+class object_range;
+class object_range_iterator;
+#endif // SIMDJSON_SUPPORTS_RANGES
+
 } // namespace ondemand
 } // namespace icelake
 } // namespace simdjson
@@ -103569,6 +130504,9 @@ template <> struct is_builtin_deserializable<icelake::ondemand::object> : std::t
 template <> struct is_builtin_deserializable<icelake::ondemand::value> : std::true_type {};
 template <> struct is_builtin_deserializable<icelake::ondemand::raw_json_string> : std::true_type {};
 template <> struct is_builtin_deserializable<std::string_view> : std::true_type {};
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template <> struct is_builtin_deserializable<std::u8string_view> : std::true_type {};
+#endif // SIMDJSON_SUPPORTS_CHAR8_T

 template <typename T>
 concept is_builtin_deserializable_v = is_builtin_deserializable<T>::value;
@@ -103586,6 +130524,10 @@ concept nothrow_custom_deserializable = nothrow_tag_invocable<deserialize_tag, V
 template <typename T, typename ValT = icelake::ondemand::value>
 concept nothrow_deserializable = nothrow_custom_deserializable<T, ValT> || is_builtin_deserializable_v<T>;

+// True iff get<T>() on ValT cannot throw: no custom tag_invoke, or that tag_invoke is noexcept.
+template <typename T, typename ValT = icelake::ondemand::value>
+concept nothrow_gettable = !custom_deserializable<T, ValT> || nothrow_custom_deserializable<T, ValT>;
+
 /// Deserialize Tag
 inline constexpr struct deserialize_tag {
   using array_type = icelake::ondemand::array;
@@ -103800,6 +130742,17 @@ public:
    */
   simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> field_key() noexcept;

+  /**
+   * Get the current field's key together with its raw byte length.
+   *
+   * Like field_key(), but also returns the number of raw key bytes (the distance
+   * from the first key byte to the closing quote). The length is recovered from
+   * the structural index -- the next structural token is the ':' -- by stepping
+   * back over any whitespace to the closing quote, avoiding a forward SIMD scan
+   * for the closing quote. Leaves the iterator positioned exactly as field_key().
+   */
+  simdjson_warn_unused simdjson_inline error_code field_key_with_length(raw_json_string &key, std::size_t &len) noexcept;
+
   /**
    * Pass the : in the field and move to its value.
    */
@@ -103952,6 +130905,8 @@ public:
   simdjson_warn_unused simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_warn_unused simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_warn_unused simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<bool> get_bool() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<bool> is_null() noexcept;
   simdjson_warn_unused simdjson_inline bool is_negative() noexcept;
@@ -103970,6 +130925,8 @@ public:
   simdjson_warn_unused simdjson_inline simdjson_result<int64_t> get_root_int64_in_string(bool check_trailing) noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<double> get_root_double(bool check_trailing) noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<double> get_root_double_in_string(bool check_trailing) noexcept;
+  simdjson_warn_unused simdjson_inline simdjson_result<float> get_root_float(bool check_trailing) noexcept;
+  simdjson_warn_unused simdjson_inline simdjson_result<float> get_root_float_in_string(bool check_trailing) noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<bool> get_root_bool(bool check_trailing) noexcept;
   simdjson_warn_unused simdjson_inline bool is_root_negative() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<bool> is_root_integer(bool check_trailing) noexcept;
@@ -104105,6 +131062,15 @@ protected:

   /** @copydoc error_code json_iterator::position() const noexcept; */
   simdjson_inline token_position position() const noexcept;
+  /**
+   * Move the live iterator directly to the given position and depth, without
+   * validating against the parser's per-depth container-start bookkeeping
+   * (unlike json_iterator::reenter_child()). Used to restore a previously
+   * captured mid-container position (see object::revert_position()): that
+   * bookkeeping only tracks each container's own start, not every position
+   * a caller might later capture and revert to, so it does not apply here.
+   */
+  simdjson_inline void reenter_at(token_position position, depth_t depth) noexcept;
   /** @copydoc error_code json_iterator::end_position() const noexcept; */
   simdjson_inline token_position last_position() const noexcept;
   /** @copydoc error_code json_iterator::end_position() const noexcept; */
@@ -104173,9 +131139,12 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+   * Supported types: object, array, raw_json_string, string_view, u8string_view (C++20), uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool
    *
-   * You may use get_double(), get_bool(), get_uint64(), get_int64(),
+   * You may use get_double(), get_float(), get_bool(), get_uint64(), get_int64(),
+   * get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(),
+   * get_float32(), get_float64(),
    * get_object(), get_array(), get_raw_json_string(), or get_string() instead.
    * When SIMDJSON_SUPPORTS_CONCEPTS is set, custom types are also supported.
    *
@@ -104185,7 +131154,7 @@ public:
   template <typename T>
   simdjson_inline simdjson_result<T> get()
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+    noexcept(nothrow_gettable<T, value>)
 #else
     noexcept
 #endif
@@ -104200,7 +131169,8 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+   * Supported types: object, array, raw_json_string, string_view, u8string_view (C++20), uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool
    * If the macro SIMDJSON_SUPPORTS_CONCEPTS is set, then custom types are also supported.
    *
    * @param out This is set to a value of the given type, parsed from the JSON. If there is an error, this may not be initialized.
@@ -104210,7 +131180,7 @@ public:
   template <typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out)
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+    noexcept(nothrow_gettable<T, value>)
 #else
     noexcept
 #endif
@@ -104238,7 +131208,7 @@ public:
       "And you do not seem to have added support for it. Indeed, we have that "
       "simdjson::custom_deserializable<T> is false and the type T is not a default type "
       "such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-      "int64_t, double, or bool.");
+      "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
     static_cast<void>(out); // to get rid of unused errors
     return UNINITIALIZED;
   }
@@ -104247,7 +131217,8 @@ public:
     // immediately fail.
     static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
       "The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-      "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+      "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+      "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
       " get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
       " You may also add support for custom types, see our documentation.");
     static_cast<void>(out); // to get rid of unused errors
@@ -104325,6 +131296,50 @@ public:
    */
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;

+  /**
+   * Cast this JSON value to a 16-bit unsigned integer.
+   *
+   * Calls get_uint64() and checks that the result fits in a uint16_t.
+   *
+   * @returns A 16-bit unsigned integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint16_t.
+   */
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+
+  /**
+   * Cast this JSON value to a 16-bit signed integer.
+   *
+   * Calls get_int64() and checks that the result fits in an int16_t.
+   *
+   * @returns A 16-bit signed integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int16_t.
+   */
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+
+  /**
+   * Cast this JSON value to an 8-bit unsigned integer.
+   *
+   * Calls get_uint64() and checks that the result fits in a uint8_t.
+   *
+   * @returns An 8-bit unsigned integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint8_t.
+   */
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+
+  /**
+   * Cast this JSON value to an 8-bit signed integer.
+   *
+   * Calls get_int64() and checks that the result fits in an int8_t.
+   *
+   * @returns An 8-bit signed integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int8_t.
+   */
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
+
   /**
    * Cast this JSON value to a double.
    *
@@ -104341,6 +131356,53 @@ public:
    */
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;

+  /**
+   * Cast this JSON value to a float (binary32).
+   *
+   * The value is rounded directly to binary32: it is the float nearest to the
+   * JSON number. Note that this may differ from get_double() followed by a cast
+   * to float, which rounds twice.
+   *
+   * @returns A float.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+
+  /**
+   * Cast this JSON value (inside string) to a float (binary32).
+   *
+   * @returns A float.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  /**
+   * Cast this JSON value to a std::float32_t (C++23).
+   *
+   * Same as get_float(): the value is rounded directly to binary32.
+   *
+   * @returns A std::float32_t.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  /**
+   * Cast this JSON value to a std::float64_t (C++23).
+   *
+   * Same as get_double().
+   *
+   * @returns A std::float64_t.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   */
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
+
   /**
    * Cast this JSON value to a string.
    *
@@ -104368,6 +131430,26 @@ public:
    */
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;

+#if SIMDJSON_SUPPORTS_CHAR8_T
+  /**
+   * Cast this JSON value to a C++20 UTF-8 string.
+   *
+   * The string is guaranteed to be valid UTF-8.
+   *
+   * Equivalent to get<std::u8string_view>(). This is a zero-copy alias of
+   * get_string(): the very same bytes are returned, viewed as char8_t.
+   *
+   * Important: a value should be consumed once. Calling get_u8string() twice on the same
+   * value is an error.
+   *
+   * @param allow_replacement Whether to allow a replacement character for unmatched surrogate pairs.
+   * @returns An UTF-8 string. The string is stored in the parser and will be invalidated the next
+   *          time it parses a document or when it is destroyed.
+   * @returns INCORRECT_TYPE if the JSON value is not a string.
+   */
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
   /**
    * Attempts to fill the provided std::string reference with the parsed value of the current string.
    *
@@ -104455,7 +131537,7 @@ public:
   /**
    * Cast this JSON value to an unsigned integer.
    *
-   * @returns A signed 64-bit integer.
+   * @returns A unsigned 64-bit integer.
    * @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit unsigned integer.
    */
   simdjson_inline operator uint64_t() noexcept(false);
@@ -104920,9 +132002,24 @@ public:
   simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template <typename string_type>
   simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -104930,9 +132027,19 @@ public:
   simdjson_inline simdjson_result<bool> get_bool() noexcept;
   simdjson_inline simdjson_result<bool> is_null() noexcept;

-  template<typename T> simdjson_inline simdjson_result<T> get() noexcept;
+  template<typename T> simdjson_inline simdjson_result<T> get()
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, icelake::ondemand::value>);
+#else
+    noexcept;
+#endif

-  template<typename T> simdjson_inline error_code get(T &out) noexcept;
+  template<typename T> simdjson_inline error_code get(T &out)
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, icelake::ondemand::value>);
+#else
+    noexcept;
+#endif

 #if SIMDJSON_EXCEPTIONS
   template <class T>
@@ -105263,6 +132370,7 @@ protected:
   token_position _position{};

   friend class json_iterator;
+  friend class document_stream;
   friend class value_iterator;
   friend class object;
   template <typename... Args>
@@ -105354,6 +132462,9 @@ protected:
    * value of this attribute.
    */
   bool _streaming{false};
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  bool _allow_incomplete_json{false};
+#endif

 public:
   simdjson_inline json_iterator() noexcept = default;
@@ -105378,6 +132489,10 @@ public:
    * start_root_array() and start_root_object().
    */
   simdjson_inline bool streaming() const noexcept;
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  simdjson_inline bool allow_incomplete_json() const noexcept;
+  simdjson_inline size_t remaining_input_length(const uint8_t *json) const noexcept;
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON

   /**
    * Get the root value iterator
@@ -106257,33 +133372,87 @@ public:
    * @param batch_size The batch size to use. MUST be larger than the largest document. The sweet
    *                   spot is cache-related: small enough to fit in cache, yet big enough to
    *                   parse as many documents as possible in one tight loop.
-   *                   Defaults to 10MB, which has been a reasonable sweet spot in our tests.
-   * @param allow_comma_separated (defaults on false) This allows a mode where the documents are
-   *                   separated by commas instead of whitespace. It comes with a performance
-   *                   penalty because the entire document is indexed at once (and the document must be
-   *                   less than 4 GB), and there is no multithreading. In this mode, the batch_size parameter
-   *                   is effectively ignored, as it is set to at least the document size.
+   *                   Defaults to 1MB, which has been a reasonable sweet spot in our tests.
+   * @param allow_comma_separated @deprecated Use stream_format::comma_delimited instead.
+   *                   When true, maps internally to stream_format::comma_delimited.
+   *                   Defaults to false.
    * @return The stream, or an error. An empty input will yield 0 documents rather than an EMPTY error. Errors:
    *         - MEMALLOC if the parser does not have enough capacity and memory allocation fails
    *         - CAPACITY if the parser does not have enough capacity and batch_size > max_capacity.
    *         - other json errors if parsing fails. You should not rely on these errors to always the same for the
    *           same document: they may vary under runtime dispatch (so they may vary depending on your system and hardware).
    */
-  inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
-  inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
-  inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
-  inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size)
+  inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size)
     the string might be automatically padded with up to SIMDJSON_PADDING whitespace characters */
-  inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
-  inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
+  inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @private An rvalue input is destroyed at the end of the full-expression, while the
+   * returned document_stream only holds a pointer to it: iterating the stream would then
+   * read freed memory. These deleted overloads also catch a std::string_view argument,
+   * which would otherwise convert implicitly to a padded_string temporary. */
+  inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size = DEFAULT_BATCH_SIZE) = delete;// unsafe
+  /** @private @overload iterate_many(const std::string &&s, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size = DEFAULT_BATCH_SIZE) = delete;// unsafe
   /** @private We do not want to allow implicit conversion from C string to std::string. */
   simdjson_result<document_stream> iterate_many(const char *buf, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept = delete;

+#ifndef SIMDJSON_DISABLE_DEPRECATED_API
+  /**
+   * @deprecated Use iterate_many with stream_format::comma_delimited instead.
+   */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @private Deleted for the same reason as iterate_many(const std::string &&s, size_t batch_size). */
+  inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size, bool allow_comma_separated) = delete;// unsafe
+  /** @private @overload iterate_many(const std::string &&s, size_t batch_size, bool allow_comma_separated) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size, bool allow_comma_separated) = delete;// unsafe
+#endif // SIMDJSON_DISABLE_DEPRECATED_API
+
+  /**
+   * Parse a stream of JSON documents with explicit format specification.
+   *
+   * @param buf The concatenated JSON documents.
+   * @param len The length of the buffer.
+   * @param batch_size The batch size to use.
+   * @param format The stream format (whitespace_delimited, json_sequence, or comma_delimited).
+   * @return A stream of documents, or an error.
+   */
+  inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(padded_string_view s, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format)
+   *
+   * The string is padded in place if needed (see simdjson::pad), as with iterate_many(std::string &s, size_t batch_size).
+   */
+  inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size, stream_format format) noexcept;
+  /** @private Deleted for the same reason as iterate_many(const std::string &&s, size_t batch_size). */
+  inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+  /** @private @overload iterate_many(const std::string &&s, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+
   /** The capacity of this parser (the largest document it can process). */
   simdjson_pure simdjson_inline size_t capacity() const noexcept;
   /** The maximum capacity of this parser (the largest document it is allowed to process). */
@@ -106411,6 +133580,7 @@ private:
   size_t _capacity{0};
   size_t _max_capacity;
   size_t _max_depth{DEFAULT_MAX_DEPTH};
+  size_t _document_len{0};
   std::unique_ptr<uint8_t[]> string_buf{};

 #if SIMDJSON_DEVELOPMENT_CHECKS
@@ -106473,8 +133643,19 @@ public:
    * Begin array iteration.
    *
    * Part of the std::iterable interface.
+   *
+   * With SIMDJSON_DEVELOPMENT_CHECKS, the iterator locks this array while it is
+   * alive, so that reentrant access (at(), count_elements(), reset(), ...) is
+   * reported as OUT_OF_ORDER_ITERATION.
+   */
+  simdjson_inline simdjson_result<array_iterator> begin() & noexcept;
+  /**
+   * Begin iteration over a temporary array, e.g., `v.get_array().begin()`.
+   *
+   * The iterator does not depend on the array instance and may outlive it, so
+   * it does not lock it.
    */
-  simdjson_inline simdjson_result<array_iterator> begin() noexcept;
+  simdjson_inline simdjson_result<array_iterator> begin() && noexcept;
   /**
    * Sentinel representing the end of the array.
    *
@@ -106605,7 +133786,7 @@ public:
    */
   template <typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out)
-     noexcept(custom_deserializable<T, array> ? nothrow_custom_deserializable<T, array> : true) {
+     noexcept(nothrow_gettable<T, array>) {
     static_assert(custom_deserializable<T, array>);
     return deserialize(*this, out);
   }
@@ -106617,7 +133798,7 @@ public:
    */
   template <typename T>
   simdjson_inline simdjson_result<T> get()
-    noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+    noexcept(nothrow_gettable<T, array>)
   {
     static_assert(std::is_default_constructible<T>::value, "The specified type is not default constructible.");
     T out{};
@@ -106673,6 +133854,10 @@ protected:
    * iter.is_alive() == false indicates iteration is complete.
    */
   value_iterator iter{};
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  bool locked{false};
+  simdjson_inline void set_locked(bool _locked) noexcept;
+#endif

   friend class value;
   friend class document;
@@ -106694,7 +133879,8 @@ public:
   simdjson_inline simdjson_result(error_code error) noexcept; ///< @private
   simdjson_inline simdjson_result() noexcept = default;

-  simdjson_inline simdjson_result<icelake::ondemand::array_iterator> begin() noexcept;
+  simdjson_inline simdjson_result<icelake::ondemand::array_iterator> begin() & noexcept;
+  simdjson_inline simdjson_result<icelake::ondemand::array_iterator> begin() && noexcept;
   simdjson_inline simdjson_result<icelake::ondemand::array_iterator> end() noexcept;
   inline simdjson_result<size_t> count_elements() & noexcept;
   inline simdjson_result<bool> is_empty() & noexcept;
@@ -106714,7 +133900,7 @@ public:
   // TODO: move this code into object-inl.h

   template<typename T>
-  simdjson_inline simdjson_result<T> get() noexcept {
+  simdjson_inline simdjson_result<T> get() noexcept(nothrow_gettable<T, icelake::ondemand::array>) {
     if (error()) { return error(); }
     if constexpr (std::is_same_v<T, icelake::ondemand::array>) {
       return first;
@@ -106722,7 +133908,7 @@ public:
     return first.get<T>();
   }
   template<typename T>
-  simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept {
+  simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept(nothrow_gettable<T, icelake::ondemand::array>) {
     if (error()) { return error(); }
     if constexpr (std::is_same_v<T, icelake::ondemand::array>) {
       out = first;
@@ -106774,6 +133960,15 @@ public:
   /** Create a new, invalid array iterator. */
   simdjson_inline array_iterator() noexcept = default;

+#if SIMDJSON_DEVELOPMENT_CHECKS
+  simdjson_inline ~array_iterator() noexcept;
+
+  simdjson_inline array_iterator(array_iterator&&) noexcept;
+  simdjson_inline array_iterator& operator=(array_iterator&&) noexcept;
+  simdjson_inline array_iterator(const array_iterator&) noexcept;
+  simdjson_inline array_iterator& operator=(const array_iterator&) noexcept;
+#endif
+
   //
   // Iterator interface
   //
@@ -106816,6 +134011,9 @@ public:
 private:
 #if SIMDJSON_DEVELOPMENT_CHECKS
    bool has_been_referenced{false};
+   array* parent{nullptr};
+
+   simdjson_inline array_iterator(const value_iterator &_iter, array* _parent) noexcept;
 #endif
   value_iterator iter{};

@@ -106915,14 +134113,14 @@ public:
   /**
    * Cast this JSON value to an unsigned integer.
    *
-   * @returns A signed 64-bit integer.
+   * @returns A unsigned 64-bit integer.
    * @returns INCORRECT_TYPE If the JSON value is not a 64-bit unsigned integer.
    */
   simdjson_inline simdjson_result<uint64_t> get_uint64() noexcept;
   /**
    * Cast this JSON value (inside string) to an unsigned integer.
    *
-   * @returns A signed 64-bit integer.
+   * @returns A unsigned 64-bit integer.
    * @returns INCORRECT_TYPE If the JSON value is not a 64-bit unsigned integer.
    */
   simdjson_inline simdjson_result<uint64_t> get_uint64_in_string() noexcept;
@@ -106960,6 +134158,46 @@ public:
    * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int32_t.
    */
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  /**
+   * Cast this JSON value to a 16-bit unsigned integer.
+   *
+   * Calls get_uint64() and checks that the result fits in a uint16_t.
+   *
+   * @returns A 16-bit unsigned integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint16_t.
+   */
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  /**
+   * Cast this JSON value to a 16-bit signed integer.
+   *
+   * Calls get_int64() and checks that the result fits in an int16_t.
+   *
+   * @returns A 16-bit signed integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int16_t.
+   */
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  /**
+   * Cast this JSON value to an 8-bit unsigned integer.
+   *
+   * Calls get_uint64() and checks that the result fits in a uint8_t.
+   *
+   * @returns An 8-bit unsigned integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint8_t.
+   */
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  /**
+   * Cast this JSON value to an 8-bit signed integer.
+   *
+   * Calls get_int64() and checks that the result fits in an int8_t.
+   *
+   * @returns An 8-bit signed integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int8_t.
+   */
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   /**
    * Cast this JSON value to a double.
    *
@@ -106975,6 +134213,53 @@ public:
    * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
    */
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+
+  /**
+   * Cast this JSON value to a float (binary32).
+   *
+   * The value is rounded directly to binary32: it is the float nearest to the
+   * JSON number. Note that this may differ from get_double() followed by a cast
+   * to float, which rounds twice.
+   *
+   * @returns A float.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+
+  /**
+   * Cast this JSON value (inside string) to a float (binary32).
+   *
+   * @returns A float.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  /**
+   * Cast this JSON value to a std::float32_t (C++23).
+   *
+   * Same as get_float(): the value is rounded directly to binary32.
+   *
+   * @returns A std::float32_t.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  /**
+   * Cast this JSON value to a std::float64_t (C++23).
+   *
+   * Same as get_double().
+   *
+   * @returns A std::float64_t.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   */
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   /**
    * Cast this JSON value to a string.
    *
@@ -106988,6 +134273,24 @@ public:
    * @returns INCORRECT_TYPE if the JSON value is not a string.
    */
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  /**
+   * Cast this JSON value to a C++20 UTF-8 string.
+   *
+   * The string is guaranteed to be valid UTF-8.
+   *
+   * Equivalent to get<std::u8string_view>(). This is a zero-copy alias of
+   * get_string(): the very same bytes are returned, viewed as char8_t.
+   *
+   * Important: Calling get_u8string() twice on the same document is an error.
+   *
+   * @param allow_replacement Whether to allow a replacement character for unmatched surrogate pairs.
+   * @returns An UTF-8 string. The string is stored in the parser and will be invalidated the next
+   *          time it parses a document or when it is destroyed.
+   * @returns INCORRECT_TYPE if the JSON value is not a string.
+   */
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   /**
    * Attempts to fill the provided std::string reference with the parsed value of the current string.
    *
@@ -107058,9 +134361,12 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool
    *
-   * You may use get_double(), get_bool(), get_uint64(), get_int64(),
+   * You may use get_double(), get_float(), get_bool(), get_uint64(), get_int64(),
+   * get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(),
+   * get_float32(), get_float64(),
    * get_object(), get_array(), get_raw_json_string(), or get_string() instead.
    *
    * @returns A value of the given type, parsed from the JSON.
@@ -107069,7 +134375,7 @@ public:
   template <typename T>
   simdjson_inline simdjson_result<T> get() &
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    noexcept(nothrow_gettable<T, document>)
 #else
     noexcept
 #endif
@@ -107092,7 +134398,7 @@ public:
   template<typename T>
   simdjson_inline simdjson_result<T> get() &&
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    noexcept(nothrow_gettable<T, document>)
 #else
     noexcept
 #endif
@@ -107104,7 +134410,8 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool, value
+   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool, value
    *
    * Be mindful that the document instance must remain in scope while you are accessing object, array and value instances.
    *
@@ -107115,7 +134422,7 @@ public:
   template<typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out) &
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    noexcept(nothrow_gettable<T, document>)
 #else
     noexcept
 #endif
@@ -107128,7 +134435,7 @@ public:
         "And you do not seem to have added support for it. Indeed, we have that "
         "simdjson::custom_deserializable<T> is false and the type T is not a default type "
         "such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-        "int64_t, double, or bool.");
+        "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
       static_cast<void>(out); // to get rid of unused errors
       return UNINITIALIZED;
     }
@@ -107137,7 +134444,8 @@ public:
     // immediately fail.
     static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
       "The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-      "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+      "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+      "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
       " get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
       " You may also add support for custom types, see our documentation.");
     static_cast<void>(out); // to get rid of unused errors
@@ -107146,7 +134454,12 @@ public:
   }

   /** @overload template<typename T> error_code get(T &out) & noexcept */
-  template<typename T> simdjson_deprecated simdjson_inline error_code get(T &out) && noexcept;
+  template<typename T> simdjson_deprecated simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, document>);
+#else
+    noexcept;
+#endif

 #if SIMDJSON_EXCEPTIONS
   /**
@@ -107180,24 +134493,24 @@ public:
   /**
    * Cast this JSON value to an unsigned integer.
    *
-   * @returns A signed 64-bit integer.
+   * @returns A unsigned 64-bit integer.
    * @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit unsigned integer.
    */
-  simdjson_inline operator uint64_t() noexcept(false);
+  explicit simdjson_inline operator uint64_t() noexcept(false);
   /**
    * Cast this JSON value to a signed integer.
    *
    * @returns A signed 64-bit integer.
    * @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit integer.
    */
-  simdjson_inline operator int64_t() noexcept(false);
+  explicit simdjson_inline operator int64_t() noexcept(false);
   /**
    * Cast this JSON value to a double.
    *
    * @returns A double.
    * @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a valid floating-point number.
    */
-  simdjson_inline operator double() noexcept(false);
+  explicit simdjson_inline operator double() noexcept(false);
   /**
    * Cast this JSON value to a string.
    *
@@ -107207,7 +134520,7 @@ public:
    *          time it parses a document or when it is destroyed.
    * @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not a string.
    */
-  simdjson_inline operator std::string_view() noexcept(false) simdjson_lifetime_bound;
+  explicit simdjson_inline operator std::string_view() noexcept(false) simdjson_lifetime_bound;
   /**
    * Cast this JSON value to a raw_json_string.
    *
@@ -107216,14 +134529,14 @@ public:
    * @returns A pointer to the raw JSON for the given string.
    * @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not a string.
    */
-  simdjson_inline operator raw_json_string() noexcept(false) simdjson_lifetime_bound;
+  explicit simdjson_inline operator raw_json_string() noexcept(false) simdjson_lifetime_bound;
   /**
    * Cast this JSON value to a bool.
    *
    * @returns A bool value.
    * @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not true or false.
    */
-  simdjson_inline operator bool() noexcept(false);
+  explicit simdjson_inline operator bool() noexcept(false);
   /**
    * Cast this JSON value to a value when the document is an object or an array.
    *
@@ -107718,9 +135031,24 @@ public:
   simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template <typename string_type>
   simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -107732,7 +135060,7 @@ public:
   template <typename T>
   simdjson_inline simdjson_result<T> get() &
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    noexcept(nothrow_gettable<T, document_reference>)
 #else
     noexcept
 #endif
@@ -107745,7 +135073,8 @@ public:
   template<typename T>
   simdjson_inline simdjson_result<T> get() &&
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    // Forwards to document::get<T>(), so the document customization decides.
+    noexcept(nothrow_gettable<T, document>)
 #else
     noexcept
 #endif
@@ -107757,7 +135086,8 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool, value
+   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool, value
    *
    * Be mindful that the document instance must remain in scope while you are accessing object, array and value instances.
    *
@@ -107768,7 +135098,7 @@ public:
   template<typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out) &
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document_reference> : true)
+    noexcept(nothrow_gettable<T, document_reference>)
 #else
     noexcept
 #endif
@@ -107781,7 +135111,7 @@ public:
         "And you do not seem to have added support for it. Indeed, we have that "
         "simdjson::custom_deserializable<T> is false and the type T is not a default type "
         "such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-        "int64_t, double, or bool.");
+        "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
       static_cast<void>(out); // to get rid of unused errors
       return UNINITIALIZED;
     }
@@ -107790,7 +135120,8 @@ public:
     // immediately fail.
     static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
       "The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-      "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+      "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+      "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
       " get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
       " You may also add support for custom types, see our documentation.");
     static_cast<void>(out); // to get rid of unused errors
@@ -107799,7 +135130,12 @@ public:
   }

   /** @overload template<typename T> error_code get(T &out) & noexcept */
-  template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+  template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, document_reference>);
+#else
+    noexcept;
+#endif
   simdjson_inline simdjson_result<std::string_view> raw_json() noexcept;
 #if SIMDJSON_STATIC_REFLECTION
   template<constevalutil::fixed_string... FieldNames, typename T>
@@ -107812,12 +135148,12 @@ public:
   explicit simdjson_inline operator T() noexcept(false);
   simdjson_inline operator array() & noexcept(false);
   simdjson_inline operator object() & noexcept(false);
-  simdjson_inline operator uint64_t() noexcept(false);
-  simdjson_inline operator int64_t() noexcept(false);
-  simdjson_inline operator double() noexcept(false);
-  simdjson_inline operator std::string_view() noexcept(false);
-  simdjson_inline operator raw_json_string() noexcept(false);
-  simdjson_inline operator bool() noexcept(false);
+  explicit simdjson_inline operator uint64_t() noexcept(false);
+  explicit simdjson_inline operator int64_t() noexcept(false);
+  explicit simdjson_inline operator double() noexcept(false);
+  explicit simdjson_inline operator std::string_view() noexcept(false);
+  explicit simdjson_inline operator raw_json_string() noexcept(false);
+  explicit simdjson_inline operator bool() noexcept(false);
   simdjson_inline operator value() noexcept(false);
 #endif
   simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -107879,9 +135215,24 @@ public:
   simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template <typename string_type>
   simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -107890,11 +135241,31 @@ public:
   simdjson_inline simdjson_result<icelake::ondemand::value> get_value() noexcept;
   simdjson_inline simdjson_result<bool> is_null() noexcept;

-  template<typename T> simdjson_inline simdjson_result<T> get() & noexcept;
-  template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() && noexcept;
+  template<typename T> simdjson_inline simdjson_result<T> get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, icelake::ondemand::document>);
+#else
+    noexcept;
+#endif
+  template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, icelake::ondemand::document>);
+#else
+    noexcept;
+#endif

-  template<typename T> simdjson_inline error_code get(T &out) & noexcept;
-  template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+  template<typename T> simdjson_inline error_code get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, icelake::ondemand::document>);
+#else
+    noexcept;
+#endif
+  template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, icelake::ondemand::document>);
+#else
+    noexcept;
+#endif
 #if SIMDJSON_EXCEPTIONS

   using icelake::implementation_simdjson_result_base<icelake::ondemand::document>::operator*;
@@ -107903,12 +135274,12 @@ public:
   explicit simdjson_inline operator T() noexcept(false);
   simdjson_inline operator icelake::ondemand::array() & noexcept(false);
   simdjson_inline operator icelake::ondemand::object() & noexcept(false);
-  simdjson_inline operator uint64_t() noexcept(false);
-  simdjson_inline operator int64_t() noexcept(false);
-  simdjson_inline operator double() noexcept(false);
-  simdjson_inline operator std::string_view() noexcept(false);
-  simdjson_inline operator icelake::ondemand::raw_json_string() noexcept(false);
-  simdjson_inline operator bool() noexcept(false);
+  explicit simdjson_inline operator uint64_t() noexcept(false);
+  explicit simdjson_inline operator int64_t() noexcept(false);
+  explicit simdjson_inline operator double() noexcept(false);
+  explicit simdjson_inline operator std::string_view() noexcept(false);
+  explicit simdjson_inline operator icelake::ondemand::raw_json_string() noexcept(false);
+  explicit simdjson_inline operator bool() noexcept(false);
   simdjson_inline operator icelake::ondemand::value() noexcept(false);
 #endif
   simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -107974,9 +135345,24 @@ public:
   simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template <typename string_type>
   simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -107985,22 +135371,42 @@ public:
   simdjson_inline simdjson_result<icelake::ondemand::value> get_value() noexcept;
   simdjson_inline simdjson_result<bool> is_null() noexcept;

-  template<typename T> simdjson_inline simdjson_result<T> get() & noexcept;
-  template<typename T> simdjson_inline simdjson_result<T> get() && noexcept;
+  template<typename T> simdjson_inline simdjson_result<T> get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, icelake::ondemand::document_reference>);
+#else
+    noexcept;
+#endif
+  template<typename T> simdjson_inline simdjson_result<T> get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, icelake::ondemand::document_reference>);
+#else
+    noexcept;
+#endif

-  template<typename T> simdjson_inline error_code get(T &out) & noexcept;
-  template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+  template<typename T> simdjson_inline error_code get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, icelake::ondemand::document_reference>);
+#else
+    noexcept;
+#endif
+  template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, icelake::ondemand::document_reference>);
+#else
+    noexcept;
+#endif
 #if SIMDJSON_EXCEPTIONS
   template <class T>
   explicit simdjson_inline operator T() noexcept(false);
   simdjson_inline operator icelake::ondemand::array() & noexcept(false);
   simdjson_inline operator icelake::ondemand::object() & noexcept(false);
-  simdjson_inline operator uint64_t() noexcept(false);
-  simdjson_inline operator int64_t() noexcept(false);
-  simdjson_inline operator double() noexcept(false);
-  simdjson_inline operator std::string_view() noexcept(false);
-  simdjson_inline operator icelake::ondemand::raw_json_string() noexcept(false);
-  simdjson_inline operator bool() noexcept(false);
+  explicit simdjson_inline operator uint64_t() noexcept(false);
+  explicit simdjson_inline operator int64_t() noexcept(false);
+  explicit simdjson_inline operator double() noexcept(false);
+  explicit simdjson_inline operator std::string_view() noexcept(false);
+  explicit simdjson_inline operator icelake::ondemand::raw_json_string() noexcept(false);
+  explicit simdjson_inline operator bool() noexcept(false);
   simdjson_inline operator icelake::ondemand::value() noexcept(false);
 #endif
   simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -108168,10 +135574,7 @@ public:
    *   }
    *   size_t truncated = stream.truncated_bytes();
    *
-   * IMPORTANT: this value is only meaningful under the conditions below. It is
-   * computed from stage-1 bookkeeping, and outside these conditions it is not
-   * merely imprecise, it is arbitrary -- it can exceed size_in_bytes() or wrap
-   * around to a huge value. Check it only when both of the following hold:
+   * IMPORTANT: this value is only meaningful under the conditions below.
    *
    *   - you iterated all the way to the end of the stream;
    *   - no document reported an error. Iteration stops at the first failed
@@ -108180,6 +135583,9 @@ public:
    * If you need to know about a truncated tail outside those conditions, track
    * it yourself from the last successful document (see iterator::current_index()
    * and iterator::source()).
+   *
+   * An empty input (zero bytes) or an input made only of white space contains
+   * no document: truncated_bytes() returns zero.
    */
   inline size_t truncated_bytes() const noexcept;

@@ -108239,7 +135645,10 @@ public:
      *
      * The returned string_view instance is simply a map to the (unparsed)
      * source string: it may thus include white-space characters and all manner
-     * of padding.
+     * of padding. It spans the whole current document, whether or not you
+     * have already accessed (part of) the document. Thus
+     * current_index() + source().size() is the offset just past the end of the
+     * current document, which is useful when reading a stream in chunks.
      *
      * This function (source()) is experimental and the usage
      * may change in future versions of simdjson: we find the API somewhat
@@ -108293,13 +135702,16 @@ private:
    * @param buf is the raw byte buffer we need to process
    * @param len is the length of the raw byte buffer in bytes
    * @param batch_size is the size of the windows (must be strictly greater or equal to the largest JSON document)
+   * @param allow_comma_separated whether to allow comma-separated documents
+   * @param format the stream format
    */
   simdjson_inline document_stream(
     ondemand::parser &parser,
     const uint8_t *buf,
     size_t len,
     size_t batch_size,
-    bool allow_comma_separated
+    bool allow_comma_separated,
+    stream_format format = stream_format::whitespace_delimited
   ) noexcept;

   /**
@@ -108333,8 +135745,23 @@ private:
    */
   inline void next() noexcept;

-  /** Move the json_iterator of the document to the location of the next document in the stream. */
+  /**
+   * Move the json_iterator of the document to the location of the next document
+   * in the stream.
+   *
+   * For formats with a document delimiter (`newline_delimited`, `json_sequence`),
+   * when the iterator is still inside the current document (`depth() > 0`), this
+   * may jump to the next delimiter instead of walking remaining structurals. That
+   * jump does not structure-validate the unread remainder.
+   */
   inline void next_document() noexcept;
+  /** Byte that ends a document under `format`, or 0 if there is none. */
+  simdjson_inline uint8_t document_delimiter() const noexcept;
+  /**
+   * Position the iterator at the first structural at or past the next
+   * `delimiter` in the current batch. Returns false if none is found.
+   */
+  simdjson_inline bool skip_to_delimiter(uint8_t delimiter) noexcept;

   /** Get the next document index. */
   inline size_t next_batch_start() const noexcept;
@@ -108348,6 +135775,7 @@ private:
   size_t len;
   size_t batch_size;
   bool allow_comma_separated;
+  stream_format format;
   /**
    * We are going to use just one document instance. The document owns
    * the json_iterator. It implies that we only ever pass a reference
@@ -108374,7 +135802,7 @@ private:
   /** The error returned from the stage 1 thread. */
   error_code stage1_thread_error{UNINITIALIZED};
   /** The thread used to run stage 1 against the next batch in the background. */
-  std::unique_ptr<stage1_worker> worker{new(std::nothrow) stage1_worker()};
+  std::unique_ptr<stage1_worker> worker{};
   /**
    * The parser used to run stage 1 in the background. Will be swapped
    * with the regular parser when finished.
@@ -108449,6 +135877,16 @@ public:
    * call it again nor can you call key().
    */
   simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  /**
+   * Get the key as a C++20 u8string_view. This is a zero-copy alias of
+   * unescaped_key(): the very same bytes are returned, viewed as char8_t.
+   *
+   * This consumes the key: once you have called unescaped_u8key(), you cannot
+   * call it again nor can you call key().
+   */
+  simdjson_inline simdjson_warn_unused simdjson_result<std::u8string_view> unescaped_u8key(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   /**
    * Get the key as a string_view (for higher speed, consider raw_key).
    * We deliberately use a more cumbersome name (unescaped_key) to force users
@@ -108481,6 +135919,16 @@ public:
    * you can safely call it repeatedly. Use unescaped_key() to get the unescaped key.
    */
   simdjson_inline std::string_view escaped_key() const noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  /**
+   * Get the key as a C++20 u8string_view. This is a zero-copy alias of
+   * escaped_key(): the very same bytes are returned, viewed as char8_t.
+   * The string is unprocessed, so it may contain escape characters
+   * (e.g., \uXXXX or \n). It does not count as a consumption of the content:
+   * you can safely call it repeatedly.
+   */
+  simdjson_inline std::u8string_view escaped_u8key() const noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   /**
    * Get the field value.
    */
@@ -108512,11 +135960,17 @@ public:
   simdjson_inline simdjson_result() noexcept = default;

   simdjson_inline simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> unescaped_u8key(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template<typename string_type>
   simdjson_inline error_code unescaped_key(string_type &receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<icelake::ondemand::raw_json_string> key() noexcept;
   simdjson_inline simdjson_result<std::string_view> key_raw_json_token() noexcept;
   simdjson_inline simdjson_result<std::string_view> escaped_key() noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> escaped_u8key() noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   simdjson_inline simdjson_result<icelake::ondemand::value> value() noexcept;
 };

@@ -108524,6 +135978,1398 @@ public:

 #endif // SIMDJSON_GENERIC_ONDEMAND_FIELD_H
 /* end file simdjson/generic/ondemand/field.h for icelake */
+/* including simdjson/generic/ondemand/key_selector.h for icelake: #include "simdjson/generic/ondemand/key_selector.h" */
+/* begin file simdjson/generic/ondemand/key_selector.h for icelake */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+#define SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #include "simdjson/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/common_defs.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/constevalutil.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/raw_json_string.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#include <array>
+#include <string>      // std::string (key_selector::describe)
+#include <string_view>
+#include <cstddef>
+#include <cstdint>
+#include <cstring>     // std::memcpy (portable unaligned window load)
+#include <utility>     // std::index_sequence (window candidate dispatch)
+#include <type_traits> // std::integral_constant (window candidate dispatch)
+
+// AArch64 only: 32-bit ARM (ARMv7) defines __ARM_NEON too, but lacks the
+// A64-only horizontal reductions (vmaxvq_u32) used below. See issue #2885.
+#if defined(__aarch64__)
+  #include <arm_neon.h>
+  #define SIMDJSON_KEY_SELECTOR_HAS_NEON 1
+#else
+  #define SIMDJSON_KEY_SELECTOR_HAS_NEON 0
+#endif
+#if defined(__SSE2__)
+  #include <emmintrin.h>
+  #define SIMDJSON_KEY_SELECTOR_HAS_SSE2 1
+#else
+  #define SIMDJSON_KEY_SELECTOR_HAS_SSE2 0
+#endif
+#if defined(__loongarch_sx)
+  #include <lsxintrin.h>
+  #define SIMDJSON_KEY_SELECTOR_HAS_LSX 1
+#else
+  #define SIMDJSON_KEY_SELECTOR_HAS_LSX 0
+#endif
+
+#if SIMDJSON_SUPPORTS_CONCEPTS
+
+namespace simdjson {
+namespace icelake {
+namespace ondemand {
+
+namespace key_selector_detail {
+
+// Since not constexpr, triggers compile-time error.
+inline void compile_time_error(const char* message) noexcept { (void)message; }
+
+// ============================================================================
+// Compile-time perfect-hash generator.
+// It scales to ~100 keys at compile time by determining association values one (position, character)
+// symbol at a time (gperf-style) instead of an exhaustive offset search, and
+// falls back to a Hash-and-Displace construction for large/awkward key sets.
+//
+// Only flat tables survive to runtime; the lookup is a few additions plus a
+// single SIMD key comparison (see match_raw below).
+// ============================================================================
+
+// Maximum number of character positions the gperf hash may combine.
+static constexpr std::size_t MAX_POSITIONS = 16;
+// Sentinel "position" meaning "the last character of the key".
+static constexpr std::size_t LAST_CHAR = std::size_t(-1);
+// Runtime-encoded sentinels (stored in uint8 tables).
+static constexpr std::uint8_t POS_LAST_CHAR = 0xFF; // positions_[i] == last char
+static constexpr std::uint8_t HD_MODE       = 0xFF; // num_positions == H&D mode
+// Flags stored in positions[2] in H&D mode to select the key-hash variant.
+static constexpr std::size_t HD_HASH_2BYTE_FLAG = 2;
+static constexpr std::size_t HD_HASH_4BYTE_FLAG = 4;
+
+constexpr std::size_t next_power_of_2(std::size_t n) noexcept {
+    if (n == 0) { return 1; }
+    std::size_t p = 1;
+    while (p < n) { p <<= 1; }
+    return p;
+}
+
+// Character at a given position (LAST_CHAR means last character), or 256 if out
+// of bounds.
+constexpr std::size_t char_at(std::string_view key, std::size_t pos) noexcept {
+    if (pos == LAST_CHAR) {
+        if (key.empty()) { return 256; }
+        return static_cast<unsigned char>(key[key.size() - 1]);
+    }
+    if (pos >= key.size()) { return 256; }
+    return static_cast<unsigned char>(key[pos]);
+}
+
+// Count key pairs that a set of positions fails to distinguish. Keys whose
+// lengths differ modulo the table size are separated by the length term in the
+// hash, so they need no position coverage.
+template <std::size_t N>
+consteval std::size_t count_undistinguished_pairs(
+    const std::array<std::string_view, N>& keys,
+    const std::size_t* positions,
+    std::size_t num_positions,
+    std::size_t modulus) {
+    std::size_t count = 0;
+    for (std::size_t i = 0; i < N; ++i) {
+        for (std::size_t j = i + 1; j < N; ++j) {
+            if (keys[i].size() % modulus != keys[j].size() % modulus) { continue; }
+            bool distinguished = false;
+            for (std::size_t p = 0; p < num_positions; ++p) {
+                if (char_at(keys[i], positions[p]) != char_at(keys[j], positions[p])) {
+                    distinguished = true;
+                    break;
+                }
+            }
+            if (!distinguished) { ++count; }
+        }
+    }
+    return count;
+}
+
+template <std::size_t N>
+consteval bool positions_distinguish(
+    const std::array<std::string_view, N>& keys,
+    const std::size_t* positions,
+    std::size_t num_positions,
+    std::size_t modulus) {
+    return count_undistinguished_pairs<N>(keys, positions, num_positions, modulus) == 0;
+}
+
+// Number of distinct (length % modulus, char_at(key, pos)) pairs at a position.
+template <std::size_t N>
+consteval std::size_t discriminating_power(
+    const std::array<std::string_view, N>& keys,
+    std::size_t pos,
+    std::size_t modulus) {
+    struct pair { std::size_t len_mod; std::size_t ch; };
+    std::array<pair, N> pairs{};
+    for (std::size_t i = 0; i < N; ++i) {
+        pairs[i] = {keys[i].size() % modulus, char_at(keys[i], pos)};
+    }
+    std::size_t count = 0;
+    for (std::size_t i = 0; i < N; ++i) {
+        bool dup = false;
+        for (std::size_t j = 0; j < i; ++j) {
+            if (pairs[i].len_mod == pairs[j].len_mod && pairs[i].ch == pairs[j].ch) {
+                dup = true;
+                break;
+            }
+        }
+        if (!dup) { ++count; }
+    }
+    return count;
+}
+
+template <std::size_t N>
+consteval std::size_t max_key_length(const std::array<std::string_view, N>& keys) {
+    std::size_t m = 0;
+    for (std::size_t i = 0; i < N; ++i) {
+        if (keys[i].size() > m) { m = keys[i].size(); }
+    }
+    return m;
+}
+
+// Bounded backtracking DFS for a minimal set of distinguishing positions.
+template <std::size_t N>
+consteval bool backtracking_search(
+    const std::array<std::string_view, N>& keys,
+    const std::size_t* candidates,
+    std::size_t num_candidates,
+    std::size_t* positions,
+    std::size_t& num_positions_out,
+    std::size_t& budget,
+    std::size_t modulus) {
+    constexpr std::size_t MAX_DEPTH = 8;
+    std::size_t breadth = num_candidates < 20 ? num_candidates : 20;
+
+    struct frame { std::size_t depth; std::size_t next_ci; std::size_t parent_count; };
+    std::array<frame, MAX_DEPTH + 1> stack{};
+    std::size_t sp = 0;
+
+    std::size_t initial_count = count_undistinguished_pairs<N>(keys, positions, 0, modulus);
+    if (budget > 0) { --budget; }
+    if (initial_count == 0) { num_positions_out = 0; return true; }
+
+    stack[0] = {0, 0, initial_count};
+
+    while (budget > 0) {
+        if (sp > MAX_DEPTH) {
+            if (sp == 0) { break; }
+            --sp;
+            ++stack[sp].next_ci;
+            continue;
+        }
+        auto& f = stack[sp];
+        if (f.next_ci >= breadth) {
+            if (sp == 0) { break; }
+            --sp;
+            ++stack[sp].next_ci;
+            continue;
+        }
+        positions[sp] = candidates[f.next_ci];
+        --budget;
+        std::size_t new_count = count_undistinguished_pairs<N>(keys, positions, sp + 1, modulus);
+        if (new_count == 0) { num_positions_out = sp + 1; return true; }
+        if (new_count < f.parent_count && sp + 1 < MAX_DEPTH) {
+            stack[sp + 1] = {sp + 1, f.next_ci + 1, new_count};
+            ++sp;
+        } else {
+            ++f.next_ci;
+        }
+    }
+    return false;
+}
+
+// Phase 1: select character positions that distinguish all colliding pairs.
+template <std::size_t N>
+consteval std::size_t select_positions(
+    const std::array<std::string_view, N>& keys,
+    std::array<std::size_t, MAX_POSITIONS>& positions,
+    std::size_t modulus) {
+    if (positions_distinguish<N>(keys, positions.data(), 0, modulus)) { return 0; }
+
+    std::size_t max_len = max_key_length(keys);
+    constexpr std::size_t MAX_CANDIDATES = 256;
+    std::array<std::size_t, MAX_CANDIDATES> candidates{};
+    std::array<std::size_t, MAX_CANDIDATES> powers{};
+    std::size_t num_candidates = 0;
+    for (std::size_t p = 0; p < max_len && num_candidates < MAX_CANDIDATES - 1; ++p) {
+        candidates[num_candidates] = p;
+        powers[num_candidates] = discriminating_power(keys, p, modulus);
+        ++num_candidates;
+    }
+    if (num_candidates < MAX_CANDIDATES) {
+        candidates[num_candidates] = LAST_CHAR;
+        powers[num_candidates] = discriminating_power(keys, LAST_CHAR, modulus);
+        ++num_candidates;
+    }
+    for (std::size_t i = 0; i < num_candidates; ++i) {
+        for (std::size_t j = i + 1; j < num_candidates; ++j) {
+            if (powers[j] > powers[i]) {
+                auto tc = candidates[i]; candidates[i] = candidates[j]; candidates[j] = tc;
+                auto tp = powers[i]; powers[i] = powers[j]; powers[j] = tp;
+            }
+        }
+    }
+
+    positions[0] = candidates[0];
+    if (positions_distinguish<N>(keys, positions.data(), 1, modulus)) { return 1; }
+
+    positions[0] = 0;
+    positions[1] = LAST_CHAR;
+    if (positions_distinguish<N>(keys, positions.data(), 2, modulus)) { return 2; }
+
+    {
+        std::size_t budget = 5000;
+        std::size_t num_found = 0;
+        if (backtracking_search<N>(keys, candidates.data(), num_candidates,
+                                   positions.data(), num_found, budget, modulus)) {
+            return num_found;
+        }
+    }
+
+    std::size_t num_pos = 0;
+    for (std::size_t ci = 0; ci < num_candidates && num_pos < MAX_POSITIONS; ++ci) {
+        bool already = false;
+        for (std::size_t p = 0; p < num_pos; ++p) {
+            if (positions[p] == candidates[ci]) { already = true; break; }
+        }
+        if (already) { continue; }
+        positions[num_pos] = candidates[ci];
+        ++num_pos;
+        if (positions_distinguish<N>(keys, positions.data(), num_pos, modulus)) { return num_pos; }
+    }
+
+    compile_time_error("Failed to find distinguishing positions for perfect hash");
+    return 0;
+}
+
+// Result of PHF computation. A max-sized slot_to_key array lets the same struct
+// type carry any chosen table size.
+template <std::size_t N>
+struct phf_result {
+    // Allow up to 8x the minimum table size. Sparser tables solve faster.
+    static constexpr std::size_t MAX_TABLE_SIZE = next_power_of_2(N) * 8;
+    std::size_t table_size{};
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso_values{};
+    std::size_t num_positions{};
+    std::array<std::size_t, MAX_POSITIONS> positions{};
+    std::array<std::size_t, MAX_TABLE_SIZE> slot_to_key{};
+};
+
+// Partition-based asso_values search (gperf-style). Determines asso_values one
+// (position, character) symbol at a time; never revisits a value. Equivalence
+// classes (keys sharing the same undetermined symbols) keep the search cheap.
+template <std::size_t N, std::size_t M>
+consteval bool try_generate_gperf(
+    const std::array<std::string_view, N>& keys,
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS>& asso_values,
+    std::size_t& num_positions,
+    std::array<std::size_t, MAX_POSITIONS>& positions,
+    std::array<std::size_t, M>& slot_to_key) {
+    num_positions = select_positions<N>(keys, positions, M);
+
+    for (std::size_t p = 0; p < MAX_POSITIONS; ++p) {
+        for (std::size_t c = 0; c < 256; ++c) { asso_values[p][c] = 0; }
+    }
+
+    if (num_positions == 0) {
+        for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+        for (std::size_t i = 0; i < N; ++i) {
+            std::size_t slot = keys[i].size() % M;
+            if (slot_to_key[slot] != N) { return false; }
+            slot_to_key[slot] = i;
+        }
+        return true;
+    }
+
+    std::array<std::array<std::size_t, MAX_POSITIONS>, N> kchars{};
+    for (std::size_t k = 0; k < N; ++k) {
+        for (std::size_t p = 0; p < num_positions; ++p) {
+            kchars[k][p] = char_at(keys[k], positions[p]);
+        }
+    }
+
+    struct sym_t { std::size_t pos; std::size_t ch; std::size_t freq; };
+    constexpr std::size_t MAX_SYMS = MAX_POSITIONS * 256;
+    std::array<sym_t, MAX_SYMS> syms{};
+    std::size_t nsyms = 0;
+    for (std::size_t p = 0; p < num_positions; ++p) {
+        std::array<std::size_t, 256> freq{};
+        for (std::size_t k = 0; k < N; ++k) {
+            std::size_t c = kchars[k][p];
+            if (c < 256) { freq[c]++; }
+        }
+        for (std::size_t c = 0; c < 256; ++c) {
+            if (freq[c] > 0) { syms[nsyms++] = {p, c, freq[c]}; }
+        }
+    }
+    for (std::size_t i = 0; i < nsyms; ++i) {
+        for (std::size_t j = i + 1; j < nsyms; ++j) {
+            if (syms[j].freq > syms[i].freq) {
+                auto tmp = syms[i]; syms[i] = syms[j]; syms[j] = tmp;
+            }
+        }
+    }
+
+    std::array<std::size_t, N> phash{};
+    for (std::size_t k = 0; k < N; ++k) { phash[k] = keys[k].size(); }
+
+    std::array<std::array<uint64_t, 256>, MAX_POSITIONS> salt{};
+    {
+        uint64_t s = 0x9e3779b97f4a7c15ULL;
+        for (std::size_t p = 0; p < num_positions; ++p) {
+            for (std::size_t c = 0; c < 256; ++c) {
+                s = s * 6364136223846793005ULL + 1442695040888963407ULL;
+                salt[p][c] = s;
+            }
+        }
+    }
+    std::array<uint64_t, N> sig{};
+    for (std::size_t k = 0; k < N; ++k) {
+        uint64_t s = 0;
+        for (std::size_t p = 0; p < num_positions; ++p) {
+            std::size_t c = kchars[k][p];
+            if (c < 256) { s ^= salt[p][c]; }
+        }
+        sig[k] = s;
+    }
+    std::array<std::size_t, N> order{};
+    for (std::size_t k = 0; k < N; ++k) { order[k] = k; }
+
+    std::array<std::size_t, M> slot_gen{};
+    std::size_t gen = 0;
+
+    std::size_t search_limit = next_power_of_2(M);
+    if (search_limit < 32) { search_limit = 32; }
+
+    for (std::size_t si = 0; si < nsyms; ++si) {
+        std::size_t sp = syms[si].pos;
+        std::size_t sc = syms[si].ch;
+
+        uint64_t sp_salt = salt[sp][sc];
+        for (std::size_t k = 0; k < N; ++k) {
+            if (kchars[k][sp] == sc) { sig[k] ^= sp_salt; }
+        }
+
+        for (std::size_t i = 1; i < N; ++i) {
+            std::size_t x = order[i];
+            uint64_t xs = sig[x];
+            std::size_t j = i;
+            while (j > 0 && sig[order[j - 1]] > xs) {
+                order[j] = order[j - 1];
+                --j;
+            }
+            order[j] = x;
+        }
+
+        bool found = false;
+        for (std::size_t v = 0; v < search_limit && !found; ++v) {
+            bool collision = false;
+            std::size_t ci = 0;
+            while (ci < N && !collision) {
+                uint64_t class_sig = sig[order[ci]];
+                std::size_t cj = ci;
+                while (cj < N && sig[order[cj]] == class_sig) { ++cj; }
+                if (cj - ci > 1) {
+                    ++gen;
+                    for (std::size_t x = ci; x < cj; ++x) {
+                        std::size_t k = order[x];
+                        std::size_t h = phash[k];
+                        if (kchars[k][sp] == sc) { h += v; }
+                        h %= M;
+                        if (slot_gen[h] == gen) { collision = true; break; }
+                        slot_gen[h] = gen;
+                    }
+                }
+                ci = cj;
+            }
+            if (!collision) {
+                asso_values[sp][sc] = v;
+                for (std::size_t k = 0; k < N; ++k) {
+                    if (kchars[k][sp] == sc) { phash[k] += v; }
+                }
+                found = true;
+            }
+        }
+        if (!found) { return false; }
+    }
+
+    for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+    for (std::size_t i = 0; i < N; ++i) {
+        std::size_t slot = phash[i] % M;
+        if (slot_to_key[slot] != N) { return false; }
+        slot_to_key[slot] = i;
+    }
+    std::size_t filled = 0;
+    for (std::size_t i = 0; i < M; ++i) {
+        if (slot_to_key[i] != N) { ++filled; }
+    }
+    return filled == N;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_compute_phf(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+    static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso{};
+    std::size_t npos{};
+    std::array<std::size_t, MAX_POSITIONS> pos{};
+    std::array<std::size_t, M> s2k{};
+    if (try_generate_gperf<N, M>(keys, asso, npos, pos, s2k)) {
+        result.table_size = M;
+        result.asso_values = asso;
+        result.num_positions = npos;
+        result.positions = pos;
+        for (std::size_t i = 0; i < M; ++i) { result.slot_to_key[i] = s2k[i]; }
+        for (std::size_t i = M; i < phf_result<N>::MAX_TABLE_SIZE; ++i) { result.slot_to_key[i] = N; }
+        return true;
+    }
+    return false;
+}
+
+template <std::size_t N, std::size_t M, std::size_t MaxM>
+consteval bool try_gperf_po2(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+    if (try_compute_phf<N, M>(keys, result)) { return true; }
+    constexpr std::size_t NextM = M * 2;
+    if constexpr (NextM <= MaxM) { return try_gperf_po2<N, NextM, MaxM>(keys, result); }
+    return false;
+}
+
+// --- Hash-and-Displace fallback --------------------------------------------
+
+constexpr std::size_t hd_bucket_hash(std::string_view key) noexcept {
+    std::size_t c0 = key.empty() ? 0 : static_cast<unsigned char>(key[0]);
+    std::size_t c1 = key.empty() ? 0 : static_cast<unsigned char>(key[key.size() - 1]);
+    return (c0 + c1 * 3 + key.size() * 17) & 0xFF;
+}
+constexpr std::size_t hd_safe_char(const char* p, std::size_t len, std::size_t idx) noexcept {
+    std::size_t has = static_cast<std::size_t>(idx < len);
+    std::size_t si = idx & (std::size_t{0} - has);
+    return static_cast<unsigned char>(p[si]) & (std::size_t{0} - has);
+}
+constexpr std::size_t hd_key_hash_2(std::string_view key) noexcept {
+    std::size_t kc = key.size();
+    kc = kc * 31 + static_cast<unsigned char>(key[0]);
+    kc = kc * 31 + hd_safe_char(key.data(), key.size(), 1);
+    return kc;
+}
+constexpr std::size_t hd_key_hash_4(std::string_view key) noexcept {
+    std::size_t kc = key.size();
+    kc = kc * 31 + static_cast<unsigned char>(key[0]);
+    kc = kc * 31 + hd_safe_char(key.data(), key.size(), 1);
+    kc = kc * 31 + hd_safe_char(key.data(), key.size(), 2);
+    kc = kc * 31 + hd_safe_char(key.data(), key.size(), 3);
+    return kc;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_hash_and_displace(
+    const std::array<std::string_view, N>& keys,
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS>& asso_values,
+    std::size_t& num_positions,
+    std::array<std::size_t, MAX_POSITIONS>& positions,
+    std::array<std::size_t, M>& slot_to_key) {
+    num_positions = select_positions<N>(keys, positions, M);
+
+    if (num_positions == 0) {
+        for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+        for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+        for (std::size_t i = 0; i < N; ++i) {
+            std::size_t slot = keys[i].size() % M;
+            if (slot_to_key[slot] != N) { return false; }
+            slot_to_key[slot] = i;
+        }
+        return true;
+    }
+
+    for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+    num_positions = HD_MODE; // sentinel for H&D mode
+    positions[0] = 0;
+    positions[1] = LAST_CHAR;
+
+    std::array<std::size_t, N> key_bucket{};
+    for (std::size_t i = 0; i < N; ++i) { key_bucket[i] = hd_bucket_hash(keys[i]); }
+
+    struct bucket_info { std::size_t ch; std::size_t count; };
+    std::array<bucket_info, N> buckets{};
+    std::size_t num_buckets = 0;
+    for (std::size_t i = 0; i < N; ++i) {
+        std::size_t bk = key_bucket[i];
+        bool found = false;
+        for (std::size_t b = 0; b < num_buckets; ++b) {
+            if (buckets[b].ch == bk) { ++buckets[b].count; found = true; break; }
+        }
+        if (!found) { buckets[num_buckets++] = {bk, 1}; }
+    }
+    for (std::size_t i = 0; i < num_buckets; ++i) {
+        for (std::size_t j = i + 1; j < num_buckets; ++j) {
+            if (buckets[j].count > buckets[i].count) {
+                auto tmp = buckets[i]; buckets[i] = buckets[j]; buckets[j] = tmp;
+            }
+        }
+    }
+
+    auto try_placement = [&](auto key_hash_fn) -> bool {
+        for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+        for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+        for (std::size_t b = 0; b < num_buckets; ++b) {
+            std::size_t ch = buckets[b].ch;
+            std::array<std::size_t, N> bucket_keys{};
+            std::size_t bk_count = 0;
+            for (std::size_t i = 0; i < N; ++i) {
+                if (key_bucket[i] == ch) { bucket_keys[bk_count++] = i; }
+            }
+            bool placed = false;
+            std::size_t max_d = M < 255 ? M : 255;
+            for (std::size_t d = 0; d < max_d; ++d) {
+                bool ok = true;
+                std::array<std::size_t, N> bucket_slots{};
+                for (std::size_t k = 0; k < bk_count; ++k) {
+                    std::size_t slot = (d + key_hash_fn(keys[bucket_keys[k]])) % M;
+                    if (slot_to_key[slot] != N) { ok = false; break; }
+                    for (std::size_t k2 = 0; k2 < k; ++k2) {
+                        if (bucket_slots[k2] == slot) { ok = false; break; }
+                    }
+                    if (!ok) { break; }
+                    bucket_slots[k] = slot;
+                }
+                if (ok) {
+                    asso_values[0][ch] = d;
+                    for (std::size_t k = 0; k < bk_count; ++k) {
+                        slot_to_key[bucket_slots[k]] = bucket_keys[k];
+                    }
+                    placed = true;
+                    break;
+                }
+            }
+            if (!placed) { return false; }
+        }
+        std::size_t filled = 0;
+        for (std::size_t i = 0; i < M; ++i) {
+            if (slot_to_key[i] != N) { ++filled; }
+        }
+        return filled == N;
+    };
+
+    if (try_placement([](std::string_view k) { return hd_key_hash_2(k); })) {
+        positions[2] = HD_HASH_2BYTE_FLAG;
+        return true;
+    }
+    if (try_placement([](std::string_view k) { return hd_key_hash_4(k); })) {
+        positions[2] = HD_HASH_4BYTE_FLAG;
+        return true;
+    }
+    return false;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_compute_phf_hd(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+    static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso{};
+    std::size_t npos{};
+    std::array<std::size_t, MAX_POSITIONS> pos{};
+    std::array<std::size_t, M> s2k{};
+    if (try_hash_and_displace<N, M>(keys, asso, npos, pos, s2k)) {
+        result.table_size = M;
+        result.asso_values = asso;
+        result.num_positions = npos;
+        result.positions = pos;
+        for (std::size_t i = 0; i < M; ++i) { result.slot_to_key[i] = s2k[i]; }
+        for (std::size_t i = M; i < phf_result<N>::MAX_TABLE_SIZE; ++i) { result.slot_to_key[i] = N; }
+        return true;
+    }
+    return false;
+}
+
+template <std::size_t N, std::size_t M>
+consteval phf_result<N> compute_phf_hd_po2(const std::array<std::string_view, N>& keys) {
+    static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+    phf_result<N> result{};
+    if (try_compute_phf_hd<N, M>(keys, result)) { return result; }
+    constexpr std::size_t NextM = M * 2;
+    if constexpr (NextM <= phf_result<N>::MAX_TABLE_SIZE) {
+        return compute_phf_hd_po2<N, NextM>(keys);
+    } else {
+        compile_time_error("Hash-and-Displace: failed to find valid table size");
+        return result;
+    }
+}
+
+// Compute a perfect hash for `keys`: try gperf at power-of-two sizes (capped so
+// the runtime tables stay within uint8 indices), then fall back to H&D.
+template <std::size_t N>
+consteval phf_result<N> compute_phf(const std::array<std::string_view, N>& keys) {
+    constexpr std::size_t StartM = next_power_of_2(N);
+    constexpr std::size_t GPERF_MAX_TABLE =
+        phf_result<N>::MAX_TABLE_SIZE < 256 ? phf_result<N>::MAX_TABLE_SIZE : 256;
+    if constexpr (StartM <= GPERF_MAX_TABLE) {
+        phf_result<N> result{};
+        if (try_gperf_po2<N, StartM, GPERF_MAX_TABLE>(keys, result)) { return result; }
+    }
+    return compute_phf_hd_po2<N, StartM>(keys);
+}
+
+// ============================================================================
+// Runtime tables (flat, uint8) derived from a phf_result.
+// ============================================================================
+
+template <std::size_t N, std::size_t TableSize, std::size_t MaxKeyLen>
+struct phf_data {
+    std::array<std::array<std::uint8_t, 256>, MAX_POSITIONS> asso_values{};
+    std::array<std::uint8_t, MAX_POSITIONS>                  positions{};
+    std::uint8_t                                             num_positions{};
+    std::uint8_t                                             hd_hash_variant{}; // 2 or 4 (H&D only)
+    std::array<std::uint8_t, TableSize>                      slot_to_key{};
+    // slot_key_bytes[s] holds the key stored at slot s, zero-padded to a 16-byte
+    // multiple so the SIMD comparison can read a whole register.
+    std::array<std::array<char, ((MaxKeyLen + 15) / 16) * 16>, TableSize> slot_key_bytes{};
+    std::array<std::uint8_t, TableSize>                      slot_key_len{};
+};
+
+template <std::size_t N>
+constexpr std::size_t compute_max_key_len(const std::array<std::string_view, N>& keys) noexcept {
+    std::size_t m = 0;
+    for (std::size_t i = 0; i < N; ++i) { if (keys[i].size() > m) { m = keys[i].size(); } }
+    return m;
+}
+
+// Validate keys and build the runtime tables from the computed perfect hash.
+template <std::size_t N, std::size_t TableSize, std::size_t MaxKeyLen>
+consteval phf_data<N, TableSize, MaxKeyLen>
+build_phf_data(const std::array<std::string_view, N>& keys, const phf_result<N>& result) {
+    for (std::size_t i = 0; i < N; ++i) {
+        if (keys[i].empty())            { compile_time_error("empty keys are not allowed in key_selector"); }
+        if (keys[i].size() > MaxKeyLen) { compile_time_error("key length exceeds MaxKeyLen"); }
+        for (char c : keys[i]) {
+            if (c == '\\') { compile_time_error("backslash not allowed in key_selector keys"); }
+            if (c == '"')  { compile_time_error("quote not allowed in key_selector keys"); }
+            if (c == '\0') { compile_time_error("null byte not allowed in key_selector keys"); }
+        }
+        for (std::size_t j = i + 1; j < N; ++j) {
+            if (keys[i] == keys[j]) { compile_time_error("duplicate keys in key_selector"); }
+        }
+    }
+
+    phf_data<N, TableSize, MaxKeyLen> out{};
+
+    if (result.num_positions == HD_MODE) {
+        // H&D mode: single displacement table in asso_values[0].
+        for (std::size_t c = 0; c < 256; ++c) {
+            out.asso_values[0][c] = static_cast<std::uint8_t>(result.asso_values[0][c]);
+        }
+        out.num_positions   = static_cast<std::uint8_t>(HD_MODE);
+        out.hd_hash_variant = static_cast<std::uint8_t>(result.positions[2]);
+    } else {
+        for (std::size_t pi = 0; pi < result.num_positions; ++pi) {
+            for (std::size_t c = 0; c < 256; ++c) {
+                out.asso_values[pi][c] = static_cast<std::uint8_t>(result.asso_values[pi][c] % TableSize);
+            }
+        }
+        out.num_positions = static_cast<std::uint8_t>(result.num_positions);
+        for (std::size_t i = 0; i < result.num_positions; ++i) {
+            out.positions[i] = (result.positions[i] == LAST_CHAR)
+                ? POS_LAST_CHAR
+                : static_cast<std::uint8_t>(result.positions[i]);
+        }
+    }
+
+    for (std::size_t s = 0; s < TableSize; ++s) {
+        out.slot_to_key[s] = static_cast<std::uint8_t>(result.slot_to_key[s]);
+    }
+
+    for (std::size_t s = 0; s < TableSize; ++s) {
+        std::size_t ki = result.slot_to_key[s];
+        if (ki < N) {
+            auto k = keys[ki];
+            out.slot_key_len[s] = static_cast<std::uint8_t>(k.size());
+            for (std::size_t c = 0; c < k.size(); ++c) { out.slot_key_bytes[s][c] = k[c]; }
+        } else {
+            out.slot_key_len[s] = 0; // empty slot: no length can match
+        }
+    }
+    return out;
+}
+
+// --- SIMD runtime primitives ------------------------------------------------
+
+// The runtime matchers read whole 16-byte blocks up to offset 63 (four blocks)
+// for keys as long as the 63-character maximum. That read must stay within the
+// buffer's trailing padding, so the padding has to exceed the largest offset we
+// touch.
+static_assert(SIMDJSON_PADDING > 63,
+              "key_selector requires SIMDJSON_PADDING > 63 for its SIMD key reads");
+
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+// True when every byte of `diff` is zero. A 32-bit-lane horizontal max is
+// enough (zero iff every 32-bit word is zero) and is cheaper than a byte-wide
+// reduction. Do not replace this with a floating-point compare against 0.0:
+// flush-to-zero would treat a denormal as zero.
+simdjson_really_inline bool neon_all_bytes_zero(uint8x16_t diff) noexcept {
+    return vmaxvq_u32(vreinterpretq_u32_u8(diff)) == 0;
+}
+#endif
+
+// Scan for the terminating '"' starting at p. Returns its byte offset (= key
+// length). Caller guarantees SIMDJSON_PADDING (== 64) bytes past the JSON buffer,
+// so reading whole 16-byte blocks up to offset 63 is always safe.
+template <std::size_t MaxKeyLen>
+simdjson_really_inline std::size_t scan_key_length(const char* p) noexcept {
+    // The closing quote of the longest key sits at most at offset MaxKeyLen, so we
+    // scan as many 16-byte blocks as it takes to cover offsets 0..MaxKeyLen. The
+    // cap (63) keeps every covered offset within the 64-byte padding guarantee, so
+    // the SIMD and scalar builds agree.
+    static_assert(MaxKeyLen <= 63, "MaxKeyLen must be <= 63 for current SIMD implementations");
+    // [[maybe_unused]]: the scalar fallback (no NEON/SSE2/LSX) does not use it.
+    [[maybe_unused]] constexpr std::size_t num_blocks = MaxKeyLen / 16 + 1;
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+    for (std::size_t b = 0; b < num_blocks; ++b) {
+        uint8x16_t v = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + b * 16);
+        uint8x16_t cmp = vceqq_u8(v, vdupq_n_u8('"'));
+        uint64_t m = vget_lane_u64(
+            vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(cmp), 4)), 0);
+        if (simdjson_likely(m != 0)) { return b * 16 + (std::size_t(__builtin_ctzll(m)) >> 2); }
+    }
+    return MaxKeyLen + 1;
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+    for (std::size_t b = 0; b < num_blocks; ++b) {
+        __m128i v   = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + b * 16));
+        __m128i cmp = _mm_cmpeq_epi8(v, _mm_set1_epi8('"'));
+        unsigned m  = static_cast<unsigned>(_mm_movemask_epi8(cmp));
+        if (simdjson_likely(m != 0)) { return b * 16 + std::size_t(__builtin_ctz(m)); }
+    }
+    return MaxKeyLen + 1;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+    for (std::size_t b = 0; b < num_blocks; ++b) {
+        __m128i v   = __lsx_vld(reinterpret_cast<const void*>(p + b * 16), 0);
+        __m128i cmp = __lsx_vseq_b(v, __lsx_vreplgr2vr_b('"'));
+        // vmskltz_b gathers the per-byte sign bits (set where the byte equals '"')
+        // into the low 16 bits of lane 0, the LSX equivalent of movemask.
+        unsigned m  = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(cmp), 0)) & 0xFFFFu;
+        if (simdjson_likely(m != 0)) { return b * 16 + std::size_t(__builtin_ctz(m)); }
+    }
+    return MaxKeyLen + 1;
+#else
+    for (std::size_t i = 0; i <= MaxKeyLen; ++i)
+        if (p[i] == '"') return i;
+    return MaxKeyLen + 1;
+#endif
+}
+
+// Byte-equal of p[0..len) against stored[0..len). stored is zero-padded past
+// `len`. Input is read over 16, 32, 48, or 64 bytes (padded JSON buffer
+// guaranteed; SIMDJSON_PADDING == 64).
+template <std::size_t MaxKeyLen>
+simdjson_really_inline bool compare_key_bytes(
+    const char* p, const char* stored, std::size_t len) noexcept {
+    // [[maybe_unused]]: only the NEON/SSE2/LSX branches read these; the scalar
+    // fallback build (no SIMD) leaves them unused, which is an error under -Werror.
+    [[maybe_unused]] alignas(16) static constexpr uint8_t idx16[16] =
+        {0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15};
+    if constexpr (MaxKeyLen <= 16) {
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+        uint8x16_t vp = vld1q_u8(reinterpret_cast<const uint8_t*>(p));
+        uint8x16_t vs = vld1q_u8(reinterpret_cast<const uint8_t*>(stored));
+        uint8x16_t mask = vcltq_u8(vld1q_u8(idx16), vdupq_n_u8(static_cast<uint8_t>(len)));
+        uint8x16_t diff = veorq_u8(vandq_u8(vp, mask), vs);
+        return neon_all_bytes_zero(diff);
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+        __m128i vp = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p));
+        __m128i vs = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored));
+        __m128i idx = _mm_load_si128(reinterpret_cast<const __m128i*>(idx16));
+        __m128i mask = _mm_cmplt_epi8(idx, _mm_set1_epi8(static_cast<char>(len)));
+        __m128i eq = _mm_cmpeq_epi8(_mm_and_si128(vp, mask), vs);
+        return _mm_movemask_epi8(eq) == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+        __m128i vp = __lsx_vld(reinterpret_cast<const void*>(p), 0);
+        __m128i vs = __lsx_vld(reinterpret_cast<const void*>(stored), 0);
+        __m128i idx = __lsx_vld(reinterpret_cast<const void*>(idx16), 0);
+        __m128i mask = __lsx_vslt_b(idx, __lsx_vreplgr2vr_b(static_cast<int>(len)));
+        __m128i eq = __lsx_vseq_b(__lsx_vand_v(vp, mask), vs);
+        return (static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq), 0)) & 0xFFFFu) == 0xFFFFu;
+#else
+        for (std::size_t i = 0; i < len; ++i)
+            if (p[i] != stored[i]) return false;
+        return true;
+#endif
+    } else if constexpr (MaxKeyLen <= 32) {
+        [[maybe_unused]] alignas(16) static constexpr uint8_t idx32_hi[16] =
+            {16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31};
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+        uint8x16_t vp_lo = vld1q_u8(reinterpret_cast<const uint8_t*>(p));
+        uint8x16_t vp_hi = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + 16);
+        uint8x16_t vs_lo = vld1q_u8(reinterpret_cast<const uint8_t*>(stored));
+        uint8x16_t vs_hi = vld1q_u8(reinterpret_cast<const uint8_t*>(stored) + 16);
+        uint8x16_t lenv = vdupq_n_u8(static_cast<uint8_t>(len));
+        uint8x16_t m_lo = vcltq_u8(vld1q_u8(idx16),    lenv);
+        uint8x16_t m_hi = vcltq_u8(vld1q_u8(idx32_hi), lenv);
+        uint8x16_t d_lo = veorq_u8(vandq_u8(vp_lo, m_lo), vs_lo);
+        uint8x16_t d_hi = veorq_u8(vandq_u8(vp_hi, m_hi), vs_hi);
+        return neon_all_bytes_zero(vorrq_u8(d_lo, d_hi));
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+        __m128i vp_lo = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p));
+        __m128i vp_hi = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + 16));
+        __m128i vs_lo = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored));
+        __m128i vs_hi = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored + 16));
+        __m128i lenv = _mm_set1_epi8(static_cast<char>(len));
+        __m128i m_lo = _mm_cmplt_epi8(_mm_load_si128(reinterpret_cast<const __m128i*>(idx16)),    lenv);
+        __m128i m_hi = _mm_cmplt_epi8(_mm_load_si128(reinterpret_cast<const __m128i*>(idx32_hi)), lenv);
+        __m128i eq_lo = _mm_cmpeq_epi8(_mm_and_si128(vp_lo, m_lo), vs_lo);
+        __m128i eq_hi = _mm_cmpeq_epi8(_mm_and_si128(vp_hi, m_hi), vs_hi);
+        return (_mm_movemask_epi8(eq_lo) & _mm_movemask_epi8(eq_hi)) == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+        __m128i lenv = __lsx_vreplgr2vr_b(static_cast<int>(len));
+        __m128i vp_lo = __lsx_vld(reinterpret_cast<const void*>(p), 0);
+        __m128i vp_hi = __lsx_vld(reinterpret_cast<const void*>(p + 16), 0);
+        __m128i vs_lo = __lsx_vld(reinterpret_cast<const void*>(stored), 0);
+        __m128i vs_hi = __lsx_vld(reinterpret_cast<const void*>(stored + 16), 0);
+        __m128i m_lo = __lsx_vslt_b(__lsx_vld(reinterpret_cast<const void*>(idx16), 0),    lenv);
+        __m128i m_hi = __lsx_vslt_b(__lsx_vld(reinterpret_cast<const void*>(idx32_hi), 0), lenv);
+        __m128i eq_lo = __lsx_vseq_b(__lsx_vand_v(vp_lo, m_lo), vs_lo);
+        __m128i eq_hi = __lsx_vseq_b(__lsx_vand_v(vp_hi, m_hi), vs_hi);
+        unsigned mlo = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq_lo), 0)) & 0xFFFFu;
+        unsigned mhi = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq_hi), 0)) & 0xFFFFu;
+        return (mlo & mhi) == 0xFFFFu;
+#else
+        for (std::size_t i = 0; i < len; ++i)
+            if (p[i] != stored[i]) return false;
+        return true;
+#endif
+    } else if constexpr (MaxKeyLen <= 64) {
+        // 3 or 4 16-byte blocks (33..48 -> 3, 49..64 -> 4). stored is zero-padded
+        // to exactly num_blocks*16 bytes (KEY_STRIDE), so neither load overruns it.
+        // [[maybe_unused]]: the scalar fallback (no NEON/SSE2/LSX) does not use it.
+        [[maybe_unused]] constexpr std::size_t num_blocks = (MaxKeyLen + 15) / 16;
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+        uint8x16_t base = vld1q_u8(idx16);
+        uint8x16_t lenv = vdupq_n_u8(static_cast<uint8_t>(len));
+        uint8x16_t acc  = vdupq_n_u8(0);
+        for (std::size_t b = 0; b < num_blocks; ++b) {
+            uint8x16_t vp   = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + b * 16);
+            uint8x16_t vs   = vld1q_u8(reinterpret_cast<const uint8_t*>(stored) + b * 16);
+            uint8x16_t idxv = vaddq_u8(base, vdupq_n_u8(static_cast<uint8_t>(b * 16)));
+            uint8x16_t mask = vcltq_u8(idxv, lenv);
+            acc = vorrq_u8(acc, veorq_u8(vandq_u8(vp, mask), vs));
+        }
+        return neon_all_bytes_zero(acc);
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+        __m128i base = _mm_load_si128(reinterpret_cast<const __m128i*>(idx16));
+        __m128i lenv = _mm_set1_epi8(static_cast<char>(len));
+        int eq = 0xFFFF;
+        for (std::size_t b = 0; b < num_blocks; ++b) {
+            __m128i vp   = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + b * 16));
+            __m128i vs   = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored + b * 16));
+            __m128i idxv = _mm_add_epi8(base, _mm_set1_epi8(static_cast<char>(b * 16)));
+            __m128i mask = _mm_cmplt_epi8(idxv, lenv);
+            eq &= _mm_movemask_epi8(_mm_cmpeq_epi8(_mm_and_si128(vp, mask), vs));
+        }
+        return eq == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+        __m128i base = __lsx_vld(reinterpret_cast<const void*>(idx16), 0);
+        __m128i lenv = __lsx_vreplgr2vr_b(static_cast<int>(len));
+        unsigned acc = 0xFFFFu;
+        for (std::size_t b = 0; b < num_blocks; ++b) {
+            __m128i vp   = __lsx_vld(reinterpret_cast<const void*>(p + b * 16), 0);
+            __m128i vs   = __lsx_vld(reinterpret_cast<const void*>(stored + b * 16), 0);
+            __m128i idxv = __lsx_vadd_b(base, __lsx_vreplgr2vr_b(static_cast<int>(b * 16)));
+            __m128i mask = __lsx_vslt_b(idxv, lenv);
+            __m128i eq   = __lsx_vseq_b(__lsx_vand_v(vp, mask), vs);
+            acc &= static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq), 0)) & 0xFFFFu;
+        }
+        return acc == 0xFFFFu;
+#else
+        for (std::size_t i = 0; i < len; ++i)
+            if (p[i] != stored[i]) return false;
+        return true;
+#endif
+    } else {
+        for (std::size_t i = 0; i < len; ++i)
+            if (p[i] != stored[i]) return false;
+        return true;
+    }
+}
+
+// --- Single 8-bit window fast path ------------------------------------------
+//
+// Many small key sets can be told apart by inspecting a *single* 8-bit window of
+// the key bytes -- and that window need not be byte-aligned. Because every JSON
+// key is terminated by a '"', the bytes at and before a key's length are well
+// defined for any key at least that long: byte i is the key character when i is
+// inside the key and the closing quote when i == len. So we read two bytes at a
+// fixed offset, extract 8 consecutive bits at a fixed intra-byte shift, and if
+// that value is distinct for every key, a 256-entry table maps it straight to a
+// candidate key. The match then needs no hash and -- in the length-free overload
+// -- no SIMD length scan: load two bytes, shift, mask, index the table, and
+// confirm the candidate with one comparison.
+//
+// Allowing an *unaligned* window (a shift of 1..7) mixes bits from two adjacent
+// bytes and discriminates key sets that no single aligned byte can. For example,
+// the partial_tweets keys {created_at,id,text,in_reply_to_status_id,user,
+// retweet_count,favorite_count} share a colliding byte at every aligned position
+// 0,1,2, yet the 8 bits starting at bit offset 2 are unique across all seven.
+// Simpler cases fall out as the shift==0 special case: {"id","screen_name"}
+// splits on byte 0, {"jo","joe"} on the quote at byte 2.
+//
+// The window is confined to the first (shortest key length + 1) bytes so the
+// two-byte read never crosses a key's closing quote into uncontrolled value
+// bytes; that final byte is the shortest key's quote.
+template <std::size_t N, std::size_t MaxKeyLen>
+struct window_data {
+    static constexpr std::size_t KEY_STRIDE = ((MaxKeyLen + 15) / 16) * 16;
+    bool                                            ok{false};
+    std::uint8_t                                    byte_offset{0}; // first byte of the 2-byte read
+    std::uint8_t                                    shift{0};       // intra-byte bit shift (0..7)
+    std::array<std::uint8_t, 256>                   window_to_key{}; // window byte -> key index, N if none
+    std::array<std::uint8_t, N>                     key_len{};
+    std::array<std::array<char, KEY_STRIDE>, N>     key_bytes{};     // zero-padded
+};
+
+// Byte seen at `idx` for key `i`: a key character when idx is inside the key, or
+// the closing '"' when idx == len (callers keep idx <= every key's length).
+template <std::size_t N>
+constexpr unsigned window_byte_at(const std::array<std::string_view, N>& keys,
+                                  std::size_t i, std::size_t idx) noexcept {
+    if (idx < keys[i].size()) { return static_cast<unsigned char>(keys[i][idx]); }
+    return static_cast<unsigned>('"');
+}
+
+// The 8-bit window value for key `i` at (byte_offset, shift): the little-endian
+// pair (byte[off], byte[off+1]) shifted right by `shift` and truncated. Matches
+// the runtime read exactly.
+template <std::size_t N>
+constexpr unsigned window_value(const std::array<std::string_view, N>& keys,
+                                std::size_t i, std::size_t byte_offset, std::size_t shift) noexcept {
+    unsigned lo = window_byte_at<N>(keys, i, byte_offset);
+    unsigned hi = window_byte_at<N>(keys, i, byte_offset + 1);
+    return ((lo | (hi << 8)) >> shift) & 0xFFu;
+}
+
+template <std::size_t N, std::size_t MaxKeyLen>
+consteval window_data<N, MaxKeyLen>
+compute_window(const std::array<std::string_view, N>& keys) {
+    window_data<N, MaxKeyLen> out{};
+
+    std::size_t min_len = keys[0].size();
+    for (std::size_t i = 1; i < N; ++i) {
+        if (keys[i].size() < min_len) { min_len = keys[i].size(); }
+    }
+
+    // Iterate windows nearest the front first (cheapest to read, smallest shift).
+    for (std::size_t off = 0; off <= min_len; ++off) {
+        for (std::size_t shift = 0; shift < 8; ++shift) {
+            // The read touches byte off, and byte off+1 when shift != 0. Both must
+            // stay within the safe region [0, min_len] (min_len is the shortest
+            // key's quote index). off <= min_len is guaranteed by the loop bound.
+            if (shift != 0 && off + 1 > min_len) { continue; }
+
+            bool distinct = true;
+            for (std::size_t i = 0; i < N && distinct; ++i) {
+                for (std::size_t j = i + 1; j < N; ++j) {
+                    if (window_value<N>(keys, i, off, shift) == window_value<N>(keys, j, off, shift)) {
+                        distinct = false;
+                        break;
+                    }
+                }
+            }
+            if (!distinct) { continue; }
+
+            out.ok          = true;
+            out.byte_offset = static_cast<std::uint8_t>(off);
+            out.shift       = static_cast<std::uint8_t>(shift);
+            for (std::size_t b = 0; b < 256; ++b) { out.window_to_key[b] = static_cast<std::uint8_t>(N); }
+            for (std::size_t i = 0; i < N; ++i) {
+                out.window_to_key[window_value<N>(keys, i, off, shift)] = static_cast<std::uint8_t>(i);
+                out.key_len[i] = static_cast<std::uint8_t>(keys[i].size());
+                for (std::size_t c = 0; c < keys[i].size(); ++c) { out.key_bytes[i][c] = keys[i][c]; }
+            }
+            return out;
+        }
+    }
+    return out; // ok == false: no single 8-bit window distinguishes the keys
+}
+
+// Read the 8-bit window at (byte_offset, shift) from a padded key pointer.
+simdjson_really_inline std::uint8_t read_window(const char* p, std::size_t byte_offset,
+                                                std::size_t shift) noexcept {
+    std::uint16_t w;
+    // Two controlled bytes (within the shortest key + its quote, hence within the
+    // padded buffer). memcpy is the portable little-endian unaligned load.
+    std::memcpy(&w, p + byte_offset, sizeof(w));
+#if defined(__BYTE_ORDER__) && (__BYTE_ORDER__ == __ORDER_BIG_ENDIAN__)
+    w = static_cast<std::uint16_t>((w >> 8) | (w << 8));
+#endif
+    return static_cast<std::uint8_t>((w >> shift) & 0xFFu);
+}
+
+// Verify a window candidate. The window table already mapped the key to the
+// single possible index `ki` (< N); here we confirm it. Folding over 0..N-1
+// turns the runtime `ki` into a compile-time index in the matching arm, so the
+// candidate's length and bytes are constants for compare_key_bytes -- the same
+// specialization the ordered find_field path gets from a CT-length
+// unsafe_is_equal, and what keeps this path competitive. The closing-quote check
+// (p[L] == '"') both confirms the key ends exactly at the candidate's length and
+// rejects a wrong-length key, so this works whether or not the caller knew len.
+template <std::size_t N, std::size_t MaxKeyLen, std::size_t... Is>
+simdjson_really_inline std::size_t
+match_window_candidate(const char* p, std::uint8_t ki,
+                       const window_data<N, MaxKeyLen>& w,
+                       std::index_sequence<Is...>) noexcept {
+  std::size_t result = N;
+  auto try_match = [&](auto Ic) {
+    constexpr std::size_t i = decltype(Ic)::value;
+    if (ki == i && p[w.key_len[i]] == '"' &&
+        key_selector_detail::compare_key_bytes<MaxKeyLen>(
+            p, w.key_bytes[i].data(), w.key_len[i])) {
+      result = i;
+    }
+  };
+  (try_match(std::integral_constant<std::size_t, Is>{}), ...);
+  return result;
+}
+
+// --- describe() string helpers (constexpr; no std::to_string, which is not) ---
+
+// Append the decimal form of v to s.
+SIMDJSON_CONSTEXPR_STRING void append_uint(std::string& s, std::size_t v) {
+    if (v == 0) { s.push_back('0'); return; }
+    char buf[20];
+    std::size_t n = 0;
+    while (v > 0) { buf[n++] = static_cast<char>('0' + (v % 10)); v /= 10; }
+    while (n > 0) { s.push_back(buf[--n]); }
+}
+
+// Append a byte as its decimal value, plus the printable character in quotes
+// when it is in the printable ASCII range (e.g. "110 ('n')").
+SIMDJSON_CONSTEXPR_STRING void append_byte(std::string& s, unsigned b) {
+    append_uint(s, b);
+    if (b >= 0x20 && b < 0x7f) {
+        s += " ('";
+        s.push_back(static_cast<char>(b));
+        s += "')";
+    }
+}
+
+} // namespace key_selector_detail
+
+/**
+ * Stateless, compile-time key selector.
+ *
+ * Usage:
+ *   using sel_t = key_selector<"id", "text", "user">;
+ *   std::size_t i = sel_t::match_raw(raw_key); // returns sel_t::size() on miss
+ *
+ * The perfect hash is built at compile time (gperf-style, with a
+ * Hash-and-Displace fallback) and only flat tables survive to runtime. All
+ * tables are static constexpr, so the lookup fully inlines.
+ *
+ * Limitations:
+ *   - Each key must be at most 63 characters long (and no longer than
+ *     SIMDJSON_PADDING). Longer keys trigger a compile-time error.
+ *   - The number of keys should be moderate. The hard limit is 255 keys;
+ *     compilation time grows with the number of keys, so prefer a few dozen at
+ *     most per selector.
+ *   - Keys must be distinct, non-empty, and free of backslash, double-quote and
+ *     null bytes (matching is done against the raw, unescaped JSON key bytes).
+ */
+template <constevalutil::fixed_string... Keys>
+struct key_selector {
+    static constexpr std::size_t N = sizeof...(Keys);
+    static_assert(N > 0,   "key_selector requires at least one key");
+    static_assert(N <= 255,"key_selector supports at most 255 keys");
+
+    static constexpr std::array<std::string_view, N> keys{ Keys.view()... };
+    static constexpr std::size_t max_key_len = key_selector_detail::compute_max_key_len<N>(keys);
+    static_assert(max_key_len <= SIMDJSON_PADDING,
+                  "key longer than SIMDJSON_PADDING is not supported");
+    // The SIMD key-length scan covers offsets 0..63 (four 16-byte blocks), which
+    // stays within the 64-byte padding guarantee. A 64-character key's closing
+    // quote would land at offset 64 and be missed on NEON/SSE2/LSX while still
+    // matching in scalar builds, so cap at 63 to keep implementations in agreement.
+    static_assert(max_key_len <= 63,
+                  "key_selector keys must be at most 63 characters long");
+
+    static constexpr auto result = key_selector_detail::compute_phf<N>(keys);
+    static constexpr std::size_t table_size = result.table_size;
+
+    static constexpr auto phf =
+        key_selector_detail::build_phf_data<N, table_size, max_key_len>(keys, result);
+
+    // Single 8-bit-window discriminator (when one exists). Detected at compile
+    // time and selected with `if constexpr` below, so the hash path is compiled
+    // out for key sets that qualify, and this is compiled out for those that do
+    // not.
+    static constexpr auto window =
+        key_selector_detail::compute_window<N, max_key_len>(keys);
+
+    static constexpr std::size_t size() noexcept { return N; }
+
+    /**
+     * Look up a JSON key whose length is already known. p must point at the first
+     * key byte (just after the opening quote) in a padded simdjson buffer, and len
+     * must be the number of raw key bytes (the distance to the closing quote).
+     * Returns the selector index in [0, N) on match, or N on miss.
+     *
+     * Prefer this overload when the caller can obtain the key length cheaply (for
+     * example, object::for_each derives it from the structural index rather than
+     * re-scanning for the closing quote).
+     */
+    static simdjson_really_inline std::size_t match_raw(const char* p, std::size_t len) noexcept {
+        if (len == 0 || len > max_key_len) { return N; }
+
+        if constexpr (window.ok) {
+            // One 8-bit window selects the only possible candidate key;
+            // match_window_candidate confirms it (bytes + closing quote). p sits
+            // in a padded buffer and the window stays within the shortest key +
+            // quote, so the two-byte read is always in bounds. len is unused here
+            // because the quote check already pins the key's end.
+            std::uint8_t ki = window.window_to_key[
+                key_selector_detail::read_window(p, window.byte_offset, window.shift)];
+            if (ki >= N) { return N; }
+            return key_selector_detail::match_window_candidate(
+                p, ki, window, std::make_index_sequence<N>{});
+        }
+
+        std::size_t slot;
+        if (phf.num_positions == key_selector_detail::HD_MODE) {
+            // Hash-and-Displace: bucket displacement + per-key hash.
+            std::string_view key(p, len);
+            std::size_t bucket = key_selector_detail::hd_bucket_hash(key);
+            std::size_t kh = (phf.hd_hash_variant == key_selector_detail::HD_HASH_2BYTE_FLAG)
+                ? key_selector_detail::hd_key_hash_2(key)
+                : key_selector_detail::hd_key_hash_4(key);
+            slot = (phf.asso_values[0][bucket] + kh) & (table_size - 1);
+        } else {
+            // gperf: h = len + sum of asso_values over the selected positions.
+            // positions / num_positions / asso_values are compile-time constants,
+            // so this loop fully unrolls. The idx < len guard mirrors the
+            // generator's char_at()-> 256 -> skip behavior for out-of-range
+            // positions (required: arbitrary positions may exceed a key's length).
+            std::size_t h = len;
+            for (std::uint8_t i = 0; i < phf.num_positions; ++i) {
+                std::uint8_t pos = phf.positions[i];
+                std::size_t idx = (pos == key_selector_detail::POS_LAST_CHAR)
+                                  ? (len - std::size_t{1})
+                                  : static_cast<std::size_t>(pos);
+                if (idx < len) {
+                    h += phf.asso_values[i][static_cast<unsigned char>(p[idx])];
+                }
+            }
+            slot = h & (table_size - 1);
+        }
+
+        std::uint8_t ki = phf.slot_to_key[slot];
+        if (ki >= N) { return N; }
+        if (phf.slot_key_len[slot] != len) { return N; }
+        if (!key_selector_detail::compare_key_bytes<max_key_len>(
+                p, phf.slot_key_bytes[slot].data(), len)) { return N; }
+        return ki;
+    }
+
+    /**
+     * Look up a JSON key. rjs must point just after an opening quote in a padded
+     * simdjson buffer. Returns the selector index in [0, N) on match, or N on miss.
+     * The key length is recovered with a SIMD scan for the closing quote; callers
+     * that already know the length should use the (p, len) overload above.
+     */
+    static simdjson_really_inline std::size_t match_raw(raw_json_string rjs) noexcept {
+        const char* p = rjs.raw();
+        if constexpr (window.ok) {
+            // One 8-bit window picks the candidate; verifying the candidate's
+            // bytes and its closing '"' confirms the full key, so the length scan
+            // is unnecessary. The window read is in bounds (padding), and the
+            // candidate length is at most max_key_len.
+            std::uint8_t ki = window.window_to_key[
+                key_selector_detail::read_window(p, window.byte_offset, window.shift)];
+            if (ki >= N) { return N; }
+            return key_selector_detail::match_window_candidate(
+                p, ki, window, std::make_index_sequence<N>{});
+        }
+        return match_raw(p, key_selector_detail::scan_key_length<max_key_len>(p));
+    }
+
+    /** Return the key text at selector index i (i in [0, N)). */
+    static constexpr std::string_view key_at(std::size_t i) noexcept {
+        return keys[i];
+    }
+
+    /**
+     * Return a complete, human-readable, multi-line description of how this
+     * selector classifies a key: which algorithm was selected at compile time
+     * (single 8-bit window, gperf-style perfect hash, or hash-and-displace), the
+     * exact bytes/positions it inspects, and the contents of the lookup tables
+     * (which window bytes or hash slots map to which key). The text mirrors what
+     * match_raw() does step by step.
+     *
+     * Everything it reports is derived from the compile-time tables, so describe()
+     * is itself usable in a constant expression when the standard library supports
+     * constexpr std::string (__cpp_lib_constexpr_string):
+     *
+     *   static_assert(!key_selector<"name", "city">::describe().empty());
+     *
+     * It allocates a std::string and is meant for documentation, debugging and
+     * tests, not for any hot path.
+     */
+    static SIMDJSON_CONSTEXPR_STRING std::string describe() {
+        std::string s;
+        s += "key_selector: ";
+        key_selector_detail::append_uint(s, N);
+        s += " keys, max key length ";
+        key_selector_detail::append_uint(s, max_key_len);
+        s += "\nkeys:\n";
+        for (std::size_t i = 0; i < N; ++i) {
+            s += "  [";
+            key_selector_detail::append_uint(s, i);
+            s += "] \"";
+            s += keys[i];
+            s += "\" (length ";
+            key_selector_detail::append_uint(s, keys[i].size());
+            s += ")\n";
+        }
+        if constexpr (window.ok) {
+            // Mirrors the window fast path of match_raw().
+            s += "algorithm: single 8-bit window\n";
+            s += "  step 1: read 2 bytes at offset ";
+            key_selector_detail::append_uint(s, window.byte_offset);
+            s += ", interpret them as a little-endian 16-bit value, shift right by ";
+            key_selector_detail::append_uint(s, window.shift);
+            s += " bits, and keep the low 8 bits\n";
+            s += "  step 2: map that byte through a 256-entry table to a key index (";
+            key_selector_detail::append_uint(s, N);
+            s += " means no match):\n";
+            for (std::size_t b = 0; b < 256; ++b) {
+                if (window.window_to_key[b] < N) {
+                    s += "    byte ";
+                    key_selector_detail::append_byte(s, static_cast<unsigned>(b));
+                    s += " -> key ";
+                    key_selector_detail::append_uint(s, window.window_to_key[b]);
+                    s += "\n";
+                }
+            }
+            s += "  step 3: confirm the candidate by checking the closing quote sits at the key's length and comparing the key bytes\n";
+        } else {
+            // Mirrors the perfect-hash path of match_raw().
+            if constexpr (phf.num_positions == key_selector_detail::HD_MODE) {
+                s += "algorithm: hash-and-displace perfect hash\n";
+                s += "  step 1: bucket = (first_byte + last_byte*3 + length*17) mod 256\n";
+                s += "  step 2: keyhash = base-31 rolling hash of the length and the first ";
+                key_selector_detail::append_uint(s, phf.hd_hash_variant);
+                s += " bytes\n";
+                s += "  step 3: slot = (displacement[bucket] + keyhash) mod ";
+                key_selector_detail::append_uint(s, table_size);
+                s += "\n  step 4: slot_to_key[slot] gives the candidate key index\n";
+                s += "  per-key derivation:\n";
+                for (std::size_t i = 0; i < N; ++i) {
+                    std::string_view k = keys[i];
+                    std::size_t bucket = key_selector_detail::hd_bucket_hash(k);
+                    std::size_t kh = (phf.hd_hash_variant == key_selector_detail::HD_HASH_2BYTE_FLAG)
+                        ? key_selector_detail::hd_key_hash_2(k)
+                        : key_selector_detail::hd_key_hash_4(k);
+                    std::size_t slot = (phf.asso_values[0][bucket] + kh) & (table_size - 1);
+                    s += "    \"";
+                    s += k;
+                    s += "\": bucket=";
+                    key_selector_detail::append_uint(s, bucket);
+                    s += " displacement=";
+                    key_selector_detail::append_uint(s, phf.asso_values[0][bucket]);
+                    s += " keyhash=";
+                    key_selector_detail::append_uint(s, kh);
+                    s += " slot=";
+                    key_selector_detail::append_uint(s, slot);
+                    s += "\n";
+                }
+            } else {
+                s += "algorithm: gperf-style perfect hash over ";
+                key_selector_detail::append_uint(s, phf.num_positions);
+                s += " character position(s)\n";
+                s += "  step 1: h = key length\n";
+                s += "  step 2: for each position below, add its association value for the key byte there (a position past the key's length contributes 0):\n";
+                for (std::size_t i = 0; i < phf.num_positions; ++i) {
+                    s += "    position ";
+                    if (phf.positions[i] == key_selector_detail::POS_LAST_CHAR) {
+                        s += "last character";
+                    } else {
+                        s += "byte index ";
+                        key_selector_detail::append_uint(s, phf.positions[i]);
+                    }
+                    s += "\n";
+                }
+                s += "  step 3: slot = h mod ";
+                key_selector_detail::append_uint(s, table_size);
+                s += " (a power of two, applied as a bitmask)\n";
+                s += "  step 4: slot_to_key[slot] gives the candidate key index\n";
+                s += "  per-key derivation:\n";
+                for (std::size_t i = 0; i < N; ++i) {
+                    std::string_view k = keys[i];
+                    std::size_t h = k.size();
+                    for (std::size_t pi = 0; pi < phf.num_positions; ++pi) {
+                        std::size_t pos = phf.positions[pi];
+                        std::size_t idx = (pos == key_selector_detail::POS_LAST_CHAR)
+                                          ? (k.size() - 1) : pos;
+                        if (idx < k.size()) {
+                            h += phf.asso_values[pi][static_cast<unsigned char>(k[idx])];
+                        }
+                    }
+                    std::size_t slot = h & (table_size - 1);
+                    s += "    \"";
+                    s += k;
+                    s += "\": h=";
+                    key_selector_detail::append_uint(s, h);
+                    s += " slot=";
+                    key_selector_detail::append_uint(s, slot);
+                    s += "\n";
+                }
+            }
+            s += "  occupied slots (slot -> key):\n";
+            for (std::size_t slot = 0; slot < table_size; ++slot) {
+                if (phf.slot_to_key[slot] < N) {
+                    s += "    slot ";
+                    key_selector_detail::append_uint(s, slot);
+                    s += " -> key ";
+                    key_selector_detail::append_uint(s, phf.slot_to_key[slot]);
+                    s += " (\"";
+                    s += keys[phf.slot_to_key[slot]];
+                    s += "\", length ";
+                    key_selector_detail::append_uint(s, phf.slot_key_len[slot]);
+                    s += ")\n";
+                }
+            }
+            s += "  confirm the candidate by checking the key length matches and comparing the key bytes\n";
+        }
+        return s;
+    }
+};
+
+namespace key_selector_detail {
+template <typename> struct is_key_selector : std::false_type {};
+template <constevalutil::fixed_string... Keys>
+struct is_key_selector<key_selector<Keys...>> : std::true_type {};
+} // namespace key_selector_detail
+
+/**
+ * Matches any instantiation of key_selector<Keys...>. Used to constrain
+ * object::for_each so that passing a non-selector type yields a clear
+ * constraint error rather than a cascade of failures inside for_each.
+ */
+template <typename T>
+concept key_selector_type = key_selector_detail::is_key_selector<T>::value;
+
+} // namespace ondemand
+} // namespace icelake
+} // namespace simdjson
+
+#endif // SIMDJSON_SUPPORTS_CONCEPTS
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+/* end file simdjson/generic/ondemand/key_selector.h for icelake */
 /* including simdjson/generic/ondemand/object.h for icelake: #include "simdjson/generic/ondemand/object.h" */
 /* begin file simdjson/generic/ondemand/object.h for icelake */
 #ifndef SIMDJSON_GENERIC_ONDEMAND_OBJECT_H
@@ -108533,6 +137379,7 @@ public:
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/implementation_simdjson_result_base.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/key_selector.h" */
 /* amalgamation skipped (editor-only): #include <vector> */
 /* amalgamation skipped (editor-only): #if SIMDJSON_STATIC_REFLECTION && SIMDJSON_SUPPORTS_CONCEPTS */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_string_builder.h"  // for constevalutil::fixed_string */
@@ -108543,6 +137390,114 @@ namespace simdjson {
 namespace icelake {
 namespace ondemand {

+#if SIMDJSON_SUPPORTS_CONCEPTS
+/**
+ * Result of object::for_each: the first error encountered (SUCCESS if none) and
+ * the number of distinct selector keys that matched during the walk. A
+ * matched_count equal to Selector::size() means every selected key was present
+ * in the object. Implicitly converts to error_code so existing callers that only
+ * care about the error (including SIMDJSON_TRY and the test ASSERT_* macros) keep
+ * working unchanged.
+ *
+ * On error paths (a handler returns a non-SUCCESS error_code, or a direct-target
+ * deserialization via value::get fails), matched_count reflects only the number
+ * of distinct selected keys that were successfully handled *before* the failure;
+ * the failing key is not counted. This is the current behavior.
+ *
+ * Marked [[nodiscard]]: for_each surfaces parse/type errors (from a matched
+ * value, a returning handler, or the structural walk itself) only through this
+ * result, so it must not be silently dropped. This matters most in builds
+ * without exceptions, where it is the only channel for those errors. To
+ * deliberately ignore it, assign to a variable or cast to void.
+ */
+struct [[nodiscard]] for_each_result {
+  error_code error{SUCCESS};
+  std::size_t matched_count{0};
+  constexpr operator error_code() const noexcept { return error; }
+};
+
+namespace key_selector_for_each_detail {
+/**
+ * A per-key handler for the variadic object::for_each is either:
+ *   - an invocable taking a value (run custom logic for that field), or
+ *   - a deserialization target T, in which case the matched value is assigned
+ *     directly via value::get(T&).
+ * The target form lets callers bind fields straight to variables without a
+ * lambda per key -- e.g. obj.for_each<"name","city","age">(name, city, age) --
+ * while still allowing a lambda in any position when custom logic is needed
+ * (for example to descend into a nested object).
+ */
+template <typename H>
+concept field_handler =
+    std::is_invocable_v<std::remove_reference_t<H>&, value> ||
+    ::simdjson::deserializable<std::remove_cvref_t<H>, value>;
+
+/**
+ * noexcept-ness of handling a single handler: nothrow-invocability for the
+ * callback form, or nothrow-deserializability for the direct-target form
+ * (builtin scalar/string targets are noexcept; custom targets follow their
+ * own tag_invoke noexcept specification).
+ */
+template <typename H>
+inline constexpr bool nothrow_field_handler_v =
+    std::is_invocable_v<std::remove_reference_t<H>&, value>
+        ? std::is_nothrow_invocable_v<std::remove_reference_t<H>&, value>
+        : ::simdjson::nothrow_deserializable<std::remove_cvref_t<H>, value>;
+
+/**
+ * nothrow_field_handler_v folded over a whole handler pack. The variadic
+ * for_each overloads spell their conditional-noexcept specifier with this
+ * single id-expression rather than an inline fold expression. MSVC (VS18)
+ * mishandles a fold-expression that appears directly in the noexcept-specifier
+ * of an out-of-line template definition: it fails to recognize the definition's
+ * specifier as matching the in-class declaration's and reports a spurious
+ * C2382 ("redefinition; different exception specifications"). Naming the fold
+ * here keeps the specifier a plain identifier, which matches reliably.
+ */
+template <typename... Handlers>
+inline constexpr bool nothrow_field_handlers_v =
+    (nothrow_field_handler_v<Handlers> && ...);
+} // namespace key_selector_for_each_detail
+#endif
+
+/**
+ * An opaque snapshot of an object's scanning position, obtained from
+ * object::get_current_position() and consumed by object::revert_position().
+ *
+ * This bundles the raw token position with the iteration depth that was
+ * live at the moment of capture: a scalar field value (e.g. a number)
+ * unwinds back to the object's own depth once fully consumed, while a
+ * compound value (array/object) not yet fully consumed stays one level
+ * deeper. The correct depth to restore to therefore depends on what was
+ * live when the snapshot was taken, not on the object itself, which is
+ * why both pieces travel together here rather than being recomputed later.
+ *
+ * The members are private and only constructible by object itself: a
+ * hand-built value here would let revert_position() jump to an arbitrary,
+ * unvalidated position and depth, and this type carries no way to check
+ * that a given snapshot still refers to a live object. Only ever pass
+ * along a value you got from get_current_position(), and only back to the
+ * same object, before that object has been reset() or the parser has
+ * iterate()d a new document: neither invalidates a snapshot in a way this
+ * type can detect.
+ */
+struct object_position {
+  /**
+   * Default-constructed so a variable can be declared and assigned later,
+   * matching e.g. document()/object(). Not a valid position to revert to.
+   */
+  simdjson_inline object_position() noexcept = default;
+
+private:
+  token_position position{};
+  depth_t depth{};
+
+  simdjson_inline object_position(token_position position_, depth_t depth_) noexcept
+    : position(position_), depth(depth_) {}
+
+  friend class object;
+};
+
 /**
  * A forward-only JSON object field iterator.
  */
@@ -108561,8 +137516,19 @@ public:
    * Using the iterator directly is also possible but error-prone and discouraged. In particular,
    * you must dereference the iterator exactly once per iteration (before calling '++').
    * Doing otherwise is unsafe and may lead to errors. You are responsible for ensuring
+   *
+   * With SIMDJSON_DEVELOPMENT_CHECKS, the iterator locks this object while it is
+   * alive, so that reentrant access (find_field(), reset(), ...) is reported as
+   * OUT_OF_ORDER_ITERATION.
+   */
+  simdjson_inline simdjson_result<object_iterator> begin() & noexcept;
+  /**
+   * Get an iterator to the start of a temporary object, e.g., `v.get_object().begin()`.
+   *
+   * The iterator does not depend on the object instance and may outlive it, so
+   * it does not lock it.
    */
-  simdjson_inline simdjson_result<object_iterator> begin() noexcept;
+  simdjson_inline simdjson_result<object_iterator> begin() && noexcept;
   simdjson_inline simdjson_result<object_iterator> end() noexcept;
   /**
    * Look up a field by name on an object (order-sensitive). By order-sensitive, we mean that
@@ -108574,10 +137540,11 @@ public:
    *
    * ```cpp
    * simdjson::ondemand::parser parser;
-   * auto obj = parser.parse(R"( { "x": 1, "y": 2, "z": 3 } )"_padded);
-   * double z = obj.find_field("z");
-   * double y = obj.find_field("y");
-   * double x = obj.find_field("x");
+   * auto json = R"( { "x": 1, "y": 2, "z": 3 } )"_padded;
+   * auto doc = parser.iterate(json);
+   * double z = doc.find_field("z");
+   * double y = doc.find_field("y");
+   * double x = doc.find_field("x");
    * ```
    * If you have multiple fields with a matching key ({"x": 1,  "x": 1}) be mindful
    * that only one field is returned.
@@ -108650,6 +137617,100 @@ public:
   /** @overload simdjson_inline simdjson_result<value> find_field_unordered(std::string_view key) & noexcept; */
   simdjson_inline simdjson_result<value> operator[](std::string_view key) && noexcept;

+#if SIMDJSON_SUPPORTS_CONCEPTS
+  /**
+   * Walk this object once and invoke on_match(selector_index, value) for each
+   * field whose key is in the compile-time key_selector Selector, in JSON order
+   * (first occurrence of a duplicate key wins). Iteration stops once all
+   * Selector::size() keys have matched or the object ends. The value is consumed
+   * in place, so this is a low-overhead way to extract a known set of fields
+   * regardless of their order in the JSON.
+   *
+   * Like other object iteration in simdjson, for_each consumes the object by
+   * advancing the underlying iterator state; after the call the same object
+   * instance should not be used for further field access or iteration.
+   *
+   * Usage:
+   *   using sel_t = ondemand::key_selector<"id", "text", "user">;
+   *   obj.for_each<sel_t>([&](std::size_t i, ondemand::value v) {
+   *     switch (i) { case 0: ...; case 1: ...; }
+   *   });
+   *
+   * Limitations (see key_selector): each key must be at most 63 characters long,
+   * and the number of keys should be moderate (hard limit 255; a handful is
+   * best, as the compile-time perfect hash may fail or slow compilation for
+   * large key sets). The keys must be distinct, non-empty, and free of backslash, double-quote and
+   * null bytes.
+   *
+   * The callback may return either void or an error_code. When it returns an
+   * error_code, the walk stops at the first non-SUCCESS result and that error is
+   * returned, which lets the callback surface value-parse errors.
+   *
+   * This function is conditionally noexcept: it is noexcept exactly when invoking
+   * the callback is noexcept. The callback runs inside this frame, so a throwing
+   * callback (e.g. one using the exception-throwing conversions like
+   * std::string_view(value) or uint64_t(value)) makes for_each potentially
+   * throwing too -- the exception propagates to the caller instead of crossing a
+   * noexcept boundary and calling std::terminate.
+   *
+   * @returns a for_each_result holding the first error encountered while walking
+   *          the object (including any error returned by the callback, SUCCESS if
+   *          none) and the number of distinct selector keys that matched. The
+   *          result converts implicitly to error_code, so callers that only need
+   *          the error can ignore the count.
+   */
+  template <typename Selector, typename Func>
+    requires key_selector_type<Selector> &&
+             std::is_invocable_v<Func&, std::size_t, value>
+  simdjson_inline for_each_result for_each(Func&& on_match)
+      noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, value>);
+
+  /**
+   * Variadic per-key form. Provide exactly one handler per key in the Selector
+   * (compiler-enforced). Handlers are processed in JSON document order for the
+   * matching keys. Each handler is either:
+   *   - a deserialization target (a variable), in which case the matched value
+   *     is assigned to it via value::get -- no lambda required; or
+   *   - an invocable taking the ondemand::value (for custom logic such as
+   *     descending into a nested object). It may return void or error_code;
+   *     returning error_code lets you surface parse/type errors.
+   * The two styles may be mixed freely, one handler per key.
+   *
+   * Example (bind fields straight to variables):
+   *   using fields = ondemand::key_selector<"name", "city", "age">;
+   *   obj.for_each<fields>(name, city, age);
+   *
+   * Example (mixing a target and a lambda):
+   *   obj.for_each<ondemand::key_selector<"id", "user">>(
+   *     id,                                          // assigned via value::get
+   *     [&](ondemand::value v){ u = read_user(v); }  // custom logic
+   *   );
+   *
+   * The index-based single-callback form (taking (size_t, value)) remains
+   * available for shared-state or more complex per-key logic.
+   */
+  template <typename Selector, typename... Handlers>
+    requires key_selector_type<Selector> &&
+             (sizeof...(Handlers) == Selector::size()) &&
+             (key_selector_for_each_detail::field_handler<Handlers> && ...)
+  simdjson_inline for_each_result for_each(Handlers&&... on_match)
+      noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
+  /**
+   * Direct-key shorthand. Equivalent to for_each<key_selector<Keys...>>(...).
+   * Lets you write the keys inline without a separate using/alias, binding each
+   * field straight to a variable (or a lambda, see the Selector form above):
+   *
+   *   obj.for_each<"name", "city", "age">(name, city, age);
+   */
+  template <constevalutil::fixed_string... Keys, typename... Handlers>
+    requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+             (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+             (key_selector_for_each_detail::field_handler<Handlers> && ...)
+  simdjson_inline for_each_result for_each(Handlers&&... on_match)
+      noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+#endif
+
   /**
    * Get the value associated with the given JSON pointer. We use the RFC 6901
    * https://tools.ietf.org/html/rfc6901 standard, interpreting the current node
@@ -108726,6 +137787,34 @@ public:
    * @returns true if the object contains some elements (not empty)
    */
   inline simdjson_result<bool> reset() & noexcept;
+  /**
+   * Get an opaque token representing the object's current scanning position.
+   * Pass it to revert_position() to return to this exact point later, without
+   * paying the cost of a full reset() and re-scan from the beginning.
+   *
+   * A typical use is an optional field that may or may not be next: capture
+   * the position, attempt find_field(), and on NO_SUCH_FIELD, revert_position()
+   * instead of reset() so that fields already consumed are not rescanned.
+   *
+   * The returned token is only valid for this object, and only until it is
+   * reset() or the parser iterate()s a new document; using it after either
+   * is undefined behavior (see object_position).
+   *
+   * @returns An opaque position token.
+   */
+  simdjson_inline object_position get_current_position() const noexcept;
+  /**
+   * Return the object's scanning position to a snapshot previously obtained
+   * from get_current_position(). Unlike reset(), this does not rescan the
+   * object from the beginning: fields before the captured position remain
+   * consumed, and scanning resumes exactly where the snapshot was captured.
+   *
+   * @param position A snapshot previously returned by get_current_position(),
+   *        for this same object.
+   * @returns SUCCESS, or OUT_OF_ORDER_ITERATION if called during active
+   *          iteration (SIMDJSON_DEVELOPMENT_CHECKS builds only).
+   */
+  simdjson_inline error_code revert_position(object_position position) noexcept;
   /**
    * This method scans the beginning of the object and checks whether the
    * object is empty.
@@ -108771,7 +137860,7 @@ public:
    */
   template <typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out)
-     noexcept(custom_deserializable<T, object> ? nothrow_custom_deserializable<T, object> : true) {
+     noexcept(nothrow_gettable<T, object>) {
     static_assert(custom_deserializable<T, object>);
     return deserialize(*this, out);
   }
@@ -108783,7 +137872,7 @@ public:
    */
   template <typename T>
   simdjson_inline simdjson_result<T> get()
-    noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+    noexcept(nothrow_gettable<T, object>)
   {
     static_assert(std::is_default_constructible<T>::value, "The specified type is not default constructible.");
     T out{};
@@ -108835,10 +137924,18 @@ protected:
   simdjson_warn_unused simdjson_inline error_code find_field_raw(const std::string_view key) noexcept;

   value_iterator iter{};
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  bool locked{false};
+  simdjson_inline void set_locked(bool _locked) noexcept;
+#endif

   friend class value;
   friend class document;
   friend struct simdjson_result<object>;
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  friend class object_iterator;
+  friend struct simdjson_result<object_iterator>;
+#endif
 };

 } // namespace ondemand
@@ -108854,7 +137951,8 @@ public:
   simdjson_inline simdjson_result(error_code error) noexcept; ///< @private
   simdjson_inline simdjson_result() noexcept = default;

-  simdjson_inline simdjson_result<icelake::ondemand::object_iterator> begin() noexcept;
+  simdjson_inline simdjson_result<icelake::ondemand::object_iterator> begin() & noexcept;
+  simdjson_inline simdjson_result<icelake::ondemand::object_iterator> begin() && noexcept;
   simdjson_inline simdjson_result<icelake::ondemand::object_iterator> end() noexcept;
   simdjson_inline simdjson_result<icelake::ondemand::value> find_field(std::string_view key) & noexcept;
   simdjson_inline simdjson_result<icelake::ondemand::value> find_field(std::string_view key) && noexcept;
@@ -108872,6 +137970,8 @@ public:
 #endif
   simdjson_inline error_code for_each_at_path_with_wildcard(std::string_view json_path, Func&& callback) noexcept;
   inline simdjson_result<bool> reset() noexcept;
+  inline simdjson_result<icelake::ondemand::object_position> get_current_position() noexcept;
+  inline error_code revert_position(icelake::ondemand::object_position position) noexcept;
   inline simdjson_result<bool> is_empty() noexcept;
   inline simdjson_result<size_t> count_fields() & noexcept;
   inline simdjson_result<std::string_view> raw_json() noexcept;
@@ -108879,7 +137979,7 @@ public:
   // TODO: move this code into object-inl.h

   template<typename T>
-  simdjson_inline simdjson_result<T> get() noexcept {
+  simdjson_inline simdjson_result<T> get() noexcept(nothrow_gettable<T, icelake::ondemand::object>) {
     if (error()) { return error(); }
     if constexpr (std::is_same_v<T, icelake::ondemand::object>) {
       return first;
@@ -108887,7 +137987,7 @@ public:
     return first.get<T>();
   }
   template<typename T>
-  simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept {
+  simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept(nothrow_gettable<T, icelake::ondemand::object>) {
     if (error()) { return error(); }
     if constexpr (std::is_same_v<T, icelake::ondemand::object>) {
       out = first;
@@ -108897,6 +137997,39 @@ public:
     return SUCCESS;
   }

+  /**
+   * Forwards to object::for_each on the underlying object, so error-code-style
+   * chains (e.g. doc["x"].get_object()) can call for_each without first
+   * extracting the object. If this result holds an error, that error is returned
+   * (with a zero match count) and the callback is not invoked. See
+   * object::for_each for the semantics.
+   */
+  template <typename Selector, typename Func>
+    requires icelake::ondemand::key_selector_type<Selector> &&
+             std::is_invocable_v<Func&, std::size_t, icelake::ondemand::value>
+  simdjson_inline icelake::ondemand::for_each_result for_each(Func&& on_match)
+      noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, icelake::ondemand::value>);
+
+  /**
+   * Forwarding overload for the variadic per-key form (targets and/or lambdas).
+   */
+  template <typename Selector, typename... Handlers>
+    requires icelake::ondemand::key_selector_type<Selector> &&
+             (sizeof...(Handlers) == Selector::size()) &&
+             (icelake::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+  simdjson_inline icelake::ondemand::for_each_result for_each(Handlers&&... on_match)
+      noexcept(icelake::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
+  /**
+   * Forwarding overload for the direct-key variadic form.
+   */
+  template <constevalutil::fixed_string... Keys, typename... Handlers>
+    requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+             (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+             (icelake::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+  simdjson_inline icelake::ondemand::for_each_result for_each(Handlers&&... on_match)
+      noexcept(icelake::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
 #if SIMDJSON_STATIC_REFLECTION
   // TODO: move this code into object-inl.h
   template<constevalutil::fixed_string... FieldNames, typename T>
@@ -108937,6 +138070,15 @@ public:
    */
   simdjson_inline object_iterator() noexcept = default;

+#if SIMDJSON_DEVELOPMENT_CHECKS
+   simdjson_inline ~object_iterator() noexcept;
+
+   simdjson_inline object_iterator(object_iterator&&) noexcept;
+   simdjson_inline object_iterator& operator=(object_iterator&&) noexcept;
+   simdjson_inline object_iterator(const object_iterator&) noexcept;
+   simdjson_inline object_iterator& operator=(const object_iterator&) noexcept;
+#endif
+
   //
   // Iterator interface
   //
@@ -108956,6 +138098,9 @@ public:
 private:
 #if SIMDJSON_DEVELOPMENT_CHECKS
    bool has_been_referenced{false};
+   object* parent{nullptr};
+
+   simdjson_inline object_iterator(const value_iterator &_iter, object* _parent) noexcept;
 #endif
   /**
    * The underlying JSON iterator.
@@ -109001,6 +138146,191 @@ public:

 #endif // SIMDJSON_GENERIC_ONDEMAND_OBJECT_ITERATOR_H
 /* end file simdjson/generic/ondemand/object_iterator.h for icelake */
+/* including simdjson/generic/ondemand/ranges.h for icelake: #include "simdjson/generic/ondemand/ranges.h" */
+/* begin file simdjson/generic/ondemand/ranges.h for icelake */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_RANGES_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_ONDEMAND_RANGES_H */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/field.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#if SIMDJSON_SUPPORTS_RANGES
+
+namespace simdjson {
+namespace icelake {
+namespace ondemand {
+
+/**
+ * A ranges-compatible iterator adapter for JSON arrays.
+ *
+ * Wraps array_iterator to satisfy std::input_iterator by providing:
+ * - const operator* (via mutable internal state)
+ * - post-increment operator
+ * - iterator_concept tag
+ *
+ * The mutable approach is standard for single-pass input iterators that
+ * read from external sources (similar to std::istream_iterator).
+ */
+class array_range_iterator {
+public:
+  using iterator_concept = std::input_iterator_tag;
+  using value_type = simdjson_result<value>;
+  using reference = simdjson_result<value>;
+  using difference_type = std::ptrdiff_t;
+
+  simdjson_inline array_range_iterator() noexcept = default;
+  simdjson_inline explicit array_range_iterator(array_iterator iter) noexcept;
+
+  /**
+   * Get the current element. Const-qualified for std::indirectly_readable;
+   * internally delegates to the mutable wrapped iterator.
+   */
+  simdjson_inline simdjson_result<value> operator*() const noexcept;
+  simdjson_inline array_range_iterator& operator++() noexcept;
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+  simdjson_inline void operator++(int) noexcept;
+SIMDJSON_POP_DISABLE_WARNINGS
+  /**
+   * Comparison delegates to array_iterator::operator==, which checks
+   * whether the underlying parser has finished the array (depth-based).
+   */
+  simdjson_inline friend bool operator==(const array_range_iterator& a,
+                                         const array_range_iterator& b) noexcept {
+    return a.iter_ == b.iter_;
+  }
+
+private:
+  mutable array_iterator iter_{};
+};
+
+/**
+ * A std::ranges::view over a JSON array.
+ *
+ * Wraps an ondemand::array and exposes begin()/end() that return
+ * array_range_iterator (satisfying std::input_iterator), enabling
+ * use with std::views::transform and other range adaptors.
+ *
+ * If the array's begin() returns an error (only possible under
+ * SIMDJSON_DEVELOPMENT_CHECKS), the range will be empty and error()
+ * will return the error code.
+ *
+ * Usage:
+ *   ondemand::parser parser;
+ *   auto doc = parser.iterate(json);
+ *   auto arr = doc.get_array().value();
+ *   for (auto elem : ondemand::get_range(arr)) { ... }
+ */
+class array_range {
+public:
+  simdjson_inline array_range() noexcept = default;
+  simdjson_inline explicit array_range(array& arr) noexcept;
+
+  simdjson_inline array_range_iterator begin() noexcept;
+  simdjson_inline array_range_iterator end() noexcept;
+
+  /** Returns SUCCESS if the range was created successfully, or the error code otherwise. */
+  simdjson_inline error_code error() const noexcept { return error_; }
+
+private:
+  array_iterator begin_{};
+  array_iterator end_{};
+  error_code error_{SUCCESS};
+};
+
+/**
+ * A ranges-compatible iterator adapter for JSON objects.
+ *
+ * Wraps object_iterator to satisfy std::input_iterator, yielding
+ * simdjson_result<field> elements (key-value pairs).
+ */
+class object_range_iterator {
+public:
+  using iterator_concept = std::input_iterator_tag;
+  using value_type = simdjson_result<field>;
+  using reference = simdjson_result<field>;
+  using difference_type = std::ptrdiff_t;
+
+  simdjson_inline object_range_iterator() noexcept = default;
+  simdjson_inline explicit object_range_iterator(object_iterator iter) noexcept;
+
+  simdjson_inline simdjson_result<field> operator*() const noexcept;
+  simdjson_inline object_range_iterator& operator++() noexcept;
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+  simdjson_inline void operator++(int) noexcept;
+SIMDJSON_POP_DISABLE_WARNINGS
+  simdjson_inline friend bool operator==(const object_range_iterator& a,
+                                         const object_range_iterator& b) noexcept {
+    return a.iter_ == b.iter_;
+  }
+
+private:
+  mutable object_iterator iter_{};
+};
+
+/**
+ * A std::ranges::view over a JSON object.
+ *
+ * Wraps an ondemand::object and exposes begin()/end() that return
+ * object_range_iterator, enabling use with range adaptors.
+ *
+ * If the object's begin() returns an error, the range will be empty
+ * and error() will return the error code.
+ */
+class object_range {
+public:
+  simdjson_inline object_range() noexcept = default;
+  simdjson_inline explicit object_range(object& obj) noexcept;
+
+  simdjson_inline object_range_iterator begin() noexcept;
+  simdjson_inline object_range_iterator end() noexcept;
+
+  /** Returns SUCCESS if the range was created successfully, or the error code otherwise. */
+  simdjson_inline error_code error() const noexcept { return error_; }
+
+private:
+  object_iterator begin_{};
+  object_iterator end_{};
+  error_code error_{SUCCESS};
+};
+
+/** Get a std::ranges compatible view over a JSON array. */
+simdjson_inline array_range get_range(array& arr) noexcept;
+
+/** Get a std::ranges compatible view over a JSON object (key-value pairs). */
+simdjson_inline object_range get_key_value_range(object& obj) noexcept;
+
+#if SIMDJSON_EXCEPTIONS
+/** Get a std::ranges compatible view, unwrapping the simdjson_result (throws on error). */
+simdjson_inline array_range get_range(simdjson_result<array> result);
+
+/** Get a std::ranges compatible view, unwrapping the simdjson_result (throws on error). */
+simdjson_inline object_range get_key_value_range(simdjson_result<object> result);
+#endif // SIMDJSON_EXCEPTIONS
+
+} // namespace ondemand
+} // namespace icelake
+} // namespace simdjson
+
+namespace std {
+namespace ranges {
+template<>
+inline constexpr bool enable_view<simdjson::icelake::ondemand::array_range> = true;
+template<>
+inline constexpr bool enable_view<simdjson::icelake::ondemand::object_range> = true;
+} // namespace ranges
+} // namespace std
+
+#endif // SIMDJSON_SUPPORTS_RANGES
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_RANGES_H
+/* end file simdjson/generic/ondemand/ranges.h for icelake */
 /* including simdjson/generic/ondemand/serialization.h for icelake: #include "simdjson/generic/ondemand/serialization.h" */
 /* begin file simdjson/generic/ondemand/serialization.h for icelake */
 #ifndef SIMDJSON_GENERIC_ONDEMAND_SERIALIZATION_H
@@ -109133,12 +138463,14 @@ inline std::ostream& operator<<(std::ostream& out, simdjson::simdjson_result<sim
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/annotations.h" */
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */

 #include <concepts>
 #include <limits>
 #if SIMDJSON_STATIC_REFLECTION
 #include <meta>
+#include <vector>
 // #include <static_reflection> // for std::define_static_string - header not available yet
 #endif

@@ -109163,10 +138495,25 @@ error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept {

 template <std::floating_point T>
 error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept {
-  double x;
-  SIMDJSON_TRY(val.get_double().get(x));
-  out = static_cast<T>(x);
-  return SUCCESS;
+  if constexpr (std::is_same_v<T, float>) {
+    // Going through binary64 and then rounding to binary32 would round twice
+    // and could produce a value that is not the float nearest to the JSON
+    // number, so we parse to binary32 directly.
+    return val.get_float().get(out);
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  } else if constexpr (std::is_same_v<T, std::float32_t>) {
+    // Same reason as float.
+    float x;
+    SIMDJSON_TRY(val.get_float().get(x));
+    out = static_cast<T>(x);
+    return SUCCESS;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+  } else {
+    double x;
+    SIMDJSON_TRY(val.get_double().get(x));
+    out = static_cast<T>(x);
+    return SUCCESS;
+  }
 }

 template <std::signed_integral T>
@@ -109202,11 +138549,59 @@ template <concepts::constructible_from_string_view T, typename ValT>
 error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(std::is_nothrow_constructible_v<T, std::string_view>) {
   std::string_view str;
   SIMDJSON_TRY(val.get_string().get(str));
-  out = T{str};
+  if constexpr (requires { out.assign(str.data(), str.size()); }) {
+    // Copy straight into out (e.g., std::string): building a temporary and
+    // move-assigning it is markedly slower.
+    out.assign(str.data(), str.size());
+  } else {
+    out = T{str};
+  }
+  return SUCCESS;
+}
+
+#if SIMDJSON_SUPPORTS_CHAR8_T
+// any C++20 char8_t string-like type (can be constructed from std::u8string_view),
+// such as std::u8string
+template <concepts::constructible_from_u8string_view T, typename ValT>
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(std::is_nothrow_constructible_v<T, std::u8string_view>) {
+  std::u8string_view str;
+  SIMDJSON_TRY(val.get_u8string().get(str));
+  if constexpr (requires { out.assign(str.data(), str.size()); }) {
+    // Copy straight into out (e.g., std::u8string), as for std::string above.
+    out.assign(str.data(), str.size());
+  } else {
+    out = T{str};
+  }
   return SUCCESS;
 }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T


+namespace details {
+// Whether to deserialize the elements of the container T directly into a new
+// element (emplace_one(out) and then get) rather than into a temporary that is
+// then moved into the container. A temporary of a trivially copyable type lives
+// in registers and is cheap to move, and value-initializing a small element in
+// place costs more than it saves (GCC zeroes it with rep stos). But moving a
+// large element, or a string (whose deserialization then needs a temporary
+// string and a move assignment of its own), is expensive.
+template <typename T>
+concept deserialize_in_place =
+    concepts::returns_reference<T> && requires(T &c) { c.pop_back(); } &&
+    !std::is_trivially_copyable_v<typename T::value_type> &&
+    (sizeof(typename T::value_type) > 32 || concepts::constructible_from_string_view<typename T::value_type>);
+
+// Removes the last element of the container on destruction while armed.
+template <typename T>
+struct pop_back_guard {
+  T &container;
+  bool armed{true};
+  ~pop_back_guard() {
+    if (armed) { container.pop_back(); }
+  }
+};
+} // namespace details
+
 /**
  * STL containers have several constructors including one that takes a single
  * size argument. Thus, some compilers (Visual Studio) will not be able to
@@ -109230,22 +138625,65 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
     SIMDJSON_TRY(val.get_array().get(arr));
   }

-  for (auto v : arr) {
-    if constexpr (concepts::returns_reference<T>) {
-      if (auto const err = v.get<value_type>().get(concepts::emplace_one(out));
-          err) {
-        // If an error occurs, the empty element that we just inserted gets
-        // removed. We're not using a temp variable because if T is a heavy
-        // type, we want the valid path to be the fast path and the slow path be
-        // the path that has errors in it.
-        if constexpr (requires { out.pop_back(); }) {
-          static_cast<void>(out.pop_back());
+  if constexpr (std::is_same_v<T, std::vector<value_type>> && !std::is_same_v<value_type, bool>) {
+    // Collect the elements in a per-thread scratch vector that keeps its
+    // capacity from call to call, then move them into out after reserving the
+    // exact size: out is allocated once instead of being regrown. A nested
+    // array of the same type finds the scratch busy and takes the paths below.
+    // Prior related work: jsonifier keeps a thread-local vector and sizes the
+    // caller's vector from that element count (parse_impl.hpp,
+    // https://github.com/nihilai-collective/Jsonifier).
+    struct scratch_space {
+      std::vector<value_type> elements{};
+      bool busy{false};
+    };
+    static thread_local scratch_space scratch;
+    if (!scratch.busy && out.empty()) {
+      struct release_scratch {
+        scratch_space &s;
+        T &out;
+        size_t parsed{0};
+        bool complete{false};
+        // On an error or an exception, out gets the elements parsed so far (as
+        // with the loops below), without allocating. Kept out of the hot path.
+        simdjson_never_inline void keep_parsed() noexcept {
+          s.elements.resize(parsed);
+          out.swap(s.elements);
         }
-        return err;
-      }
-    } else {
+        ~release_scratch() {
+          if (simdjson_unlikely(!complete)) { keep_parsed(); }
+          s.elements.clear();
+          // Do not hold on to the memory of a very large array.
+          if (s.elements.capacity() * sizeof(value_type) > (1 << 20)) { std::vector<value_type>().swap(s.elements); }
+          s.busy = false;
+        }
+      } release{scratch, out};
+      scratch.busy = true;
+      for (auto v : arr) {
+        SIMDJSON_TRY(v.get<value_type>(scratch.elements.emplace_back()));
+        release.parsed++;
+      }
+      out.reserve(release.parsed);
+      release.complete = true;
+      for (auto &e : scratch.elements) { out.emplace_back(std::move(e)); }
+      return SUCCESS;
+    }
+  }
+  if constexpr (details::deserialize_in_place<T>) {
+    for (auto v : arr) {
+      auto &slot = concepts::emplace_one(out);
+      // An error or an exception (a user tag_invoke may throw) must not leave
+      // a partially deserialized element behind.
+      details::pop_back_guard<T> guard{out};
+      SIMDJSON_TRY(v.get<value_type>(slot));
+      guard.armed = false;
+    }
+  } else {
+    for (auto v : arr) {
+      // Deserialize into a temporary first: an error or an exception (a user
+      // tag_invoke may throw) must not leave a default-constructed element behind.
       value_type temp;
-      if (auto const err = v.get<value_type>().get(temp); err) {
+      if (auto const err = v.get<value_type>(temp); err) {
         return err;
       }
       concepts::emplace_one(out, std::move(temp));
@@ -109286,7 +138724,7 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
 }

 template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, icelake::ondemand::object &obj, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, icelake::ondemand::object &obj, T &out) noexcept(false) {
   using value_type = typename std::remove_cvref_t<T>::mapped_type;

   out.clear();
@@ -109305,21 +138743,21 @@ error_code tag_invoke(deserialize_tag, icelake::ondemand::object &obj, T &out) n
 }

 template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, icelake::ondemand::value &val, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, icelake::ondemand::value &val, T &out) noexcept(false) {
   icelake::ondemand::object obj;
   SIMDJSON_TRY(val.get_object().get(obj));
   return simdjson::deserialize(obj, out);
 }

 template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, icelake::ondemand::document &doc, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, icelake::ondemand::document &doc, T &out) noexcept(false) {
   icelake::ondemand::object obj;
   SIMDJSON_TRY(doc.get_object().get(obj));
   return simdjson::deserialize(obj, out);
 }

 template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, icelake::ondemand::document_reference &doc, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, icelake::ondemand::document_reference &doc, T &out) noexcept(false) {
   icelake::ondemand::object obj;
   SIMDJSON_TRY(doc.get_object().get(obj));
   return simdjson::deserialize(obj, out);
@@ -109330,10 +138768,6 @@ error_code tag_invoke(deserialize_tag, icelake::ondemand::document_reference &do
  * This CPO (Customization Point Object) will help deserialize into
  * smart pointers.
  *
- * If constructing T is nothrow, this conversion should be nothrow as well since
- * we return MEMALLOC if we're not able to allocate memory instead of throwing
- * the error message.
- *
  * @tparam T The type inside the smart pointer
  * @tparam ValT document/value type
  * @param val document/value
@@ -109341,7 +138775,7 @@ error_code tag_invoke(deserialize_tag, icelake::ondemand::document_reference &do
  * @return status of the conversion
  */
 template <concepts::smart_pointer T, typename ValT>
-error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(nothrow_deserializable<typename std::remove_cvref_t<T>::element_type, ValT>) {
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
   using element_type = typename std::remove_cvref_t<T>::element_type;

   // For better error messages, don't use these as constraints on
@@ -109353,12 +138787,13 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(nothrow_deser
       std::is_default_constructible_v<element_type>,
       "The specified type inside the unique_ptr must default constructible.");

-  auto ptr = new (std::nothrow) element_type();
-  if (ptr == nullptr) {
+  // Own the allocation before get(): a user tag_invoke may throw.
+  std::unique_ptr<element_type> ptr(new (std::nothrow) element_type());
+  if (!ptr) {
     return MEMALLOC;
   }
   SIMDJSON_TRY(val.template get<element_type>(*ptr));
-  out.reset(ptr);
+  out = std::move(ptr);
   return SUCCESS;
 }

@@ -109390,53 +138825,595 @@ error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept(nothrow_deser

 template <typename T>
 constexpr bool user_defined_type = (std::is_class_v<T>
-&& !std::is_same_v<T, std::string> && !std::is_same_v<T, std::string_view> && !concepts::optional_type<T> &&
-!concepts::appendable_containers<T>);
+&& !std::is_same_v<T, std::string> && !std::is_same_v<T, std::string_view>
+#if SIMDJSON_SUPPORTS_CHAR8_T
+// The char8_t string types are class types with no reflectable members, so
+// without this they would be taken for user structs and the reflection
+// overload below would compete with the u8 string overload in
+// std_deserialize.h, making every tag_invoke call on them ambiguous.
+&& !std::is_same_v<T, std::u8string> && !std::is_same_v<T, std::u8string_view>
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+&& !concepts::optional_type<T> &&
+!concepts::appendable_containers<T>
+// simdjson's own types (array, object, value, raw_json_string, number,
+// document, document_reference) have dedicated get<T>() specializations and
+// must never go through reflection.
+&& !is_builtin_deserializable_v<T>
+&& !std::is_same_v<T, icelake::ondemand::number>
+&& !std::is_same_v<T, icelake::ondemand::document>
+&& !std::is_same_v<T, icelake::ondemand::document_reference>);
+
+
+// key_selector_reflection_detail is defined unconditionally (it only requires
+// static reflection). It provides the compile-time machinery for building a
+// key_selector from a struct's members, the per-member helpers shared by every
+// deserialization path (they implement the annotations of annotations.h), the
+// ordered per-member path used by the opt-out build (see
+// deserialize_struct_ordered below), and the scan used for structs annotated
+// with deny_unknown_fields and as an automatic fallback (see
+// deserialize_struct_scan and keys_fit_selector below).
+namespace key_selector_reflection_detail {
+
+// A member participates if it is public, non-const, and not annotated with skip
+// or skip_deserializing.
+consteval bool is_eligible_member(std::meta::info mem) {
+  return !std::meta::is_const(mem) && std::meta::is_public(mem)
+      && !simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_tag)
+      && !simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_deserializing_tag);
+}
+
+// A field is a data member reached from T through a chain of members: usually a
+// direct member of T (a path of length one), or a member of a member annotated
+// with flatten (the members of a flattened structure are fields of the enclosing
+// one). A field is identified by the type member_path<m1, m2, ..., leaf>.
+template <std::meta::info First, std::meta::info... Rest>
+struct member_path {
+  // The data member holding the value; its annotations drive (de)serialization.
+  static constexpr std::meta::info leaf = [] {
+    std::meta::info members[] = {First, Rest...};
+    return members[sizeof...(Rest)];
+  }();
+  template <typename T>
+  static simdjson_inline constexpr auto &get(T &obj) noexcept {
+    if constexpr (sizeof...(Rest) == 0) {
+      return obj.[:First:];
+    } else {
+      return member_path<Rest...>::get(obj.[:First:]);
+    }
+  }
+};
+
+// True when T can be flattened: a structure deserialized member by member (not
+// a string, a container, an optional, a smart pointer, ...).
+template <typename T>
+constexpr bool flattenable_type = user_defined_type<T> && !concepts::string_view_keyed_map<T>
+    && !concepts::container_but_not_string<T> && !concepts::smart_pointer<T>;
+
+consteval void append_eligible_fields(std::meta::info type, std::vector<std::meta::info> &prefix,
+                                      std::vector<std::meta::info> &fields) {
+  for (std::meta::info mem : std::meta::nonstatic_data_members_of(type, std::meta::access_context::unchecked())) {
+    if (!is_eligible_member(mem)) { continue; }
+    prefix.push_back(std::meta::reflect_constant(mem));
+    if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::flatten_tag)) {
+      std::meta::info flattened = simdjson::detail::flattened_type(mem);
+      if (!std::meta::extract<bool>(std::meta::substitute(^^flattenable_type, {flattened}))) {
+        throw std::meta::exception(u8"simdjson::flatten requires a member whose type is a structure deserialized member by member", mem);
+      }
+      append_eligible_fields(flattened, prefix, fields);
+    } else {
+      fields.push_back(std::meta::substitute(^^member_path, prefix));
+    }
+    prefix.pop_back();
+  }
+}
+
+// The fields of `type` that participate in deserialization, in declaration order
+// (the fields of a flattened member take its place). A field's position in this
+// list is its "field index" below.
+consteval std::vector<std::meta::info> eligible_fields(std::meta::info type) {
+  std::vector<std::meta::info> prefix;
+  std::vector<std::meta::info> fields;
+  append_eligible_fields(type, prefix, fields);
+  return fields;
+}
+
+// The data member at the end of a field path.
+consteval std::meta::info field_leaf(std::meta::info path) {
+  return std::meta::extract<std::meta::info>(std::meta::template_arguments_of(path).back());
+}
+
+// Number of fields that participate in deserialization. A class can have zero
+// eligible fields (e.g. std::chrono::time_point, whose only data member is
+// private): an empty key_selector cannot be built, so the tag_invoke below
+// special-cases this count.
+template <typename T>
+consteval std::size_t eligible_field_count() {
+  return eligible_fields(^^T).size();
+}
+
+// The keys accepted for a member (or an enumerator): its JSON key followed by its
+// aliases. They are static strings, so they can be used as template arguments.
+consteval std::vector<const char *> accepted_keys_of(std::meta::info entity) {
+  std::vector<const char *> keys;
+  for (std::string_view key : simdjson::detail::json_key_names(entity)) {
+    bool repeated = false;
+    for (const char *previous : keys) { repeated = repeated || std::string_view(previous) == key; }
+    if (!repeated) { keys.push_back(std::define_static_string(key)); }
+  }
+  return keys;
+}
+
+// Every key accepted for T: the eligible fields in order, each followed by its
+// aliases. This is the key list of T's key_selector.
+consteval std::vector<const char *> accepted_keys(std::meta::info type) {
+  std::vector<const char *> keys;
+  for (std::meta::info path : eligible_fields(type)) {
+    for (const char *key : accepted_keys_of(field_leaf(path))) { keys.push_back(key); }
+  }
+  return keys;
+}
+
+// The field index of each entry of accepted_keys(type).
+consteval std::vector<std::size_t> accepted_key_fields(std::meta::info type) {
+  std::vector<std::size_t> key_fields;
+  std::vector<std::meta::info> fields = eligible_fields(type);
+  for (std::size_t i = 0; i < fields.size(); ++i) {
+    for (std::size_t j = 0; j < accepted_keys_of(field_leaf(fields[i])).size(); ++j) { key_fields.push_back(i); }
+  }
+  return key_fields;
+}
+
+// True when no two eligible fields of T accept the same key. Otherwise one JSON
+// key would have to fill several members: this is reported at compile time.
+template <typename T>
+consteval bool accepted_keys_are_distinct() {
+  std::vector<const char *> keys = accepted_keys(^^T);
+  for (std::size_t i = 0; i < keys.size(); ++i) {
+    for (std::size_t j = i + 1; j < keys.size(); ++j) {
+      if (std::string_view(keys[i]) == std::string_view(keys[j])) { return false; }
+    }
+  }
+  return true;
+}
+
+// True when some accepted key of T is written with escape sequences in JSON (a
+// double quote, a backslash or a control character). Such a key can only be
+// matched by comparing unescaped keys (deserialize_struct_scan): obj[key] and
+// the key_selector compare the raw bytes.
+template <typename T>
+consteval bool keys_need_unescaping() {
+  for (std::string_view key : accepted_keys(^^T)) {
+    for (char c : key) {
+      if (c == '\\' || c == '"' || static_cast<unsigned char>(c) < 0x20) { return true; }
+    }
+  }
+  return false;
+}

+// True when some eligible field of T has aliases: several selector keys may then
+// map to the same field.
+template <typename T>
+consteval bool has_aliases() {
+  return accepted_keys(^^T).size() != eligible_fields(^^T).size();
+}
+
+// True when `mem` is annotated with default_value or default_from (directly, or
+// through a default_value annotation on the enclosing structure).
+template <auto mem>
+consteval bool has_default() {
+  return simdjson::detail::has_annotation(mem, ^^simdjson::detail::default_value_tag)
+      || simdjson::detail::has_annotation(std::meta::parent_of(mem), ^^simdjson::detail::default_value_tag)
+      || simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::default_from_t) != std::meta::info{};
+}
+
+// True when a missing key for `mem` is not an error: optional members and
+// members with a default.
+template <auto mem>
+consteval bool may_be_absent() {
+  return concepts::optional_type<typename [: std::meta::type_of(mem) :]> || has_default<mem>();
+}
+
+// True when every eligible field of T is required (none may be absent). In that
+// case presence can be checked with a single match count instead of a per-field
+// "seen" array.
+template <typename T>
+consteval bool all_eligible_fields_required() {
+  bool all_required = true;
+  template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+    if constexpr (may_be_absent<[: path :]::leaf>()) {
+      all_required = false;
+    }
+  }
+  return all_required;
+}
+
+// `key` as a constevalutil::fixed_string usable as an NTTP.
+template <const char *key>
+consteval auto key_fixed_string() {
+  constexpr std::string_view key_view{ key };
+  char buffer[key_view.size() + 1] = {};
+  for (std::size_t i = 0; i < key_view.size(); ++i) { buffer[i] = key_view[i]; }
+  return constevalutil::fixed_string<key_view.size() + 1>(buffer);
+}
+
+// key_selector template arguments (one fixed_string per accepted key), in the
+// order of accepted_keys.
+template <typename T>
+consteval std::vector<std::meta::info> selector_key_args() {
+  std::vector<std::meta::info> args;
+  template for (constexpr const char *key : std::define_static_array(accepted_keys(^^T))) {
+    args.push_back(std::meta::reflect_constant(key_fixed_string<key>()));
+  }
+  return args;
+}
+
+// key_selector whose keys are exactly T's accepted keys (index i <-> i-th key).
+template <typename T>
+using selector_for = typename [: std::meta::substitute(
+    ^^icelake::ondemand::key_selector, selector_key_args<T>()) :];
+
+// True when T's accepted keys satisfy every key_selector requirement, so a
+// key_selector can be built for T without a compile-time error. This mirrors the
+// key_selector limits (see key_selector.h): at most 255 keys, each key non-empty
+// and at most 63 characters, no backslash / double-quote / null byte, and all
+// keys distinct. Keys with any other control character are excluded too: they
+// are escaped in JSON, so their raw bytes never match. When this returns false
+// the deserializer falls back to deserialize_struct_scan instead of failing to
+// compile.
+template <typename T>
+consteval bool keys_fit_selector() {
+  std::vector<std::string_view> keys;
+  for (const char *key : accepted_keys(^^T)) { keys.push_back(key); }
+  if (keys.size() > 255) { return false; }
+  for (std::size_t i = 0; i < keys.size(); ++i) {
+    if (keys[i].empty() || keys[i].size() > 63) { return false; }
+    for (char c : keys[i]) {
+      if (c == '\\' || c == '"' || static_cast<unsigned char>(c) < 0x20) { return false; }
+    }
+    for (std::size_t j = i + 1; j < keys.size(); ++j) {
+      if (keys[i] == keys[j]) { return false; }
+    }
+  }
+  return true;
+}
+
+// True when the class `adapter` declares a member named deserialize.
+consteval bool declares_deserialize(std::meta::info adapter) {
+  for (std::meta::info m : std::meta::members_of(adapter, std::meta::access_context::unchecked())) {
+    if (std::meta::has_identifier(m) && std::meta::identifier_of(m) == "deserialize") { return true; }
+  }
+  return false;
+}
+
+// Deserialize a JSON value into `target`, the storage of member `mem`, through
+// the member's with<Adapter> annotation when the adapter provides a deserialize
+// function.
+template <auto mem, typename ValueT, typename M>
+simdjson_warn_unused simdjson_inline error_code deserialize_member_value(ValueT &field_value, M &target) noexcept(false) {
+  constexpr std::meta::info with_type = simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::with_t);
+  if constexpr (with_type != std::meta::info{}) {
+    using adapter = typename [: with_type :]::adapter;
+    using ondemand_value = icelake::ondemand::value;
+    if constexpr (requires { { adapter::deserialize(field_value, target) } -> std::convertible_to<error_code>; }) {
+      return adapter::deserialize(field_value, target);
+    } else if constexpr (requires(ondemand_value &v) { { adapter::deserialize(v, target) } -> std::convertible_to<error_code>; }
+                         && requires { field_value.get_value(); }) {
+      // A transparent structure read from a document: the adapter takes an
+      // ondemand::value. A scalar document cannot be viewed as a value, so it
+      // reports SCALAR_DOCUMENT_AS_VALUE (an adapter taking auto& receives the
+      // document itself and has no such limitation).
+      ondemand_value v;
+      SIMDJSON_TRY(field_value.get_value().get(v));
+      return adapter::deserialize(v, target);
+    } else {
+      static_assert(!declares_deserialize(^^adapter),
+                    "the deserialize function of a simdjson::with adapter must be callable as "
+                    "Adapter::deserialize(simdjson::ondemand::value &, T &) and return an error_code");
+      return field_value.get(target);
+    }
+  } else {
+    return field_value.get(target);
+  }
+}
+
+// Deserialize the storage `target` of member `mem` from a JSON value.
+template <auto mem, typename ValueT, typename M>
+simdjson_warn_unused simdjson_inline error_code deserialize_member(ValueT &field_value, M &target) noexcept(false) {
+  if constexpr (has_default<mem>() && std::is_default_constructible_v<M> && std::is_move_assignable_v<M>) {
+    // A present key replaces the default value: deserialize into a fresh
+    // temporary so that, e.g., a container does not append to its default
+    // content, and a failure leaves the default untouched.
+    M value{};
+    SIMDJSON_TRY(deserialize_member_value<mem>(field_value, value));
+    target = std::move(value);
+    return SUCCESS;
+  } else {
+    return deserialize_member_value<mem>(field_value, target);
+  }
+}

+// Deserialize the field with the given field index.
+template <typename T>
+simdjson_warn_unused simdjson_inline error_code deserialize_field_at(
+    std::size_t field_index, icelake::ondemand::value field_value, T &out) noexcept(false) {
+  std::size_t counter = 0;
+  template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+    using field = [: path :];
+    if (field_index == counter) { return deserialize_member<field::leaf>(field_value, field::get(out)); }
+    ++counter;
+  }
+  return SUCCESS;
+}
+
+// Called when the key(s) of member `mem`, stored in `target`, are missing from
+// the JSON object: an error for a required member, a call to the default_from
+// factory, or nothing at all.
+template <auto mem, typename M>
+simdjson_warn_unused simdjson_inline error_code on_missing_member(M &target) noexcept(false) {
+  constexpr std::meta::info default_from_type = simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::default_from_t);
+  if constexpr (default_from_type != std::meta::info{}) {
+    target = [: default_from_type :]::factory();
+    return SUCCESS;
+  } else if constexpr (may_be_absent<mem>()) {
+    // For optional and default_value members, a missing key is not an error:
+    // leave the member at its current (default) value.
+    (void)target;
+    return SUCCESS;
+  } else {
+    (void)target;
+    return NO_SUCH_FIELD;
+  }
+}
+
+// Report or handle every field that was not seen in the JSON object.
+template <typename T, std::size_t N>
+simdjson_warn_unused simdjson_inline error_code handle_missing_fields(
+    const std::array<bool, N> &seen_field, T &out) noexcept(false) {
+  std::size_t counter = 0;
+  template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+    using field = [: path :];
+    if (!seen_field[counter]) { SIMDJSON_TRY(on_missing_member<field::leaf>(field::get(out))); }
+    ++counter;
+  }
+  return SUCCESS;
+}
+
+// Ordered, per-field deserialization: one obj[key] lookup per eligible field
+// (and per alias, until one is found). This is the opt-out path
+// (-DSIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION=1), except for structs with keys
+// that need unescaping (see keys_need_unescaping).
+template <typename T>
+simdjson_warn_unused error_code deserialize_struct_ordered(
+    icelake::ondemand::object &obj, T &out) noexcept(false) {
+  template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+    using field = [: path :];
+    icelake::ondemand::value field_value;
+    error_code error = NO_SUCH_FIELD;
+    template for (constexpr const char *key : std::define_static_array(accepted_keys_of(field::leaf))) {
+      if (error == NO_SUCH_FIELD) { error = obj[std::string_view(key)].get(field_value); }
+    }
+    if (error == NO_SUCH_FIELD) {
+      SIMDJSON_TRY(on_missing_member<field::leaf>(field::get(out)));
+    } else if (error) {
+      return error;
+    } else {
+      SIMDJSON_TRY(deserialize_member<field::leaf>(field_value, field::get(out)));
+    }
+  }
+  return SUCCESS;
+}
+
+// Appends the JSON keys that serialization writes for members of `type` that
+// deserialization cannot assign (const or non-public members, directly or
+// through flatten). `all` is true inside a flattened member that is itself
+// unassignable. Keys of skip_deserializing members are not included: they are
+// unknown keys, as documented.
+consteval void append_unassignable_keys(std::meta::info type, bool all, std::vector<const char *> &keys) {
+  for (std::meta::info mem : std::meta::nonstatic_data_members_of(type, std::meta::access_context::unchecked())) {
+    if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_tag)
+        || simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_serializing_tag)
+        || simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_deserializing_tag)) {
+      continue;
+    }
+    bool unassignable = all || !is_eligible_member(mem);
+    if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::flatten_tag)) {
+      append_unassignable_keys(simdjson::detail::flattened_type(mem), unassignable, keys);
+    } else if (unassignable) {
+      keys.push_back(std::define_static_string(simdjson::detail::json_key_name(mem)));
+    }
+  }
+}
+
+consteval std::vector<const char *> unassignable_keys(std::meta::info type) {
+  std::vector<const char *> keys;
+  append_unassignable_keys(type, false, keys);
+  return keys;
+}
+
+// Deserialization by a single pass over every field of the object, comparing
+// unescaped keys. It is used for structs annotated with deny_unknown_fields
+// (DenyUnknown = true), where a key that does not map to an eligible field is
+// reported as UNKNOWN_FIELD, and as the fallback for structs whose keys the
+// key_selector or obj[key] cannot match (see keys_fit_selector and
+// keys_need_unescaping). As with object::for_each, the first occurrence of a
+// field wins (later duplicates, or aliases of a field already seen, are ignored).
+template <bool DenyUnknown, typename T>
+simdjson_warn_unused error_code deserialize_struct_scan(
+    icelake::ondemand::object &obj, T &out) noexcept(false) {
+  static constexpr auto keys = std::define_static_array(accepted_keys(^^T));
+  static constexpr auto key_fields = std::define_static_array(accepted_key_fields(^^T));
+  std::array<bool, eligible_field_count<T>()> seen_field{};
+  for (auto field_result : obj) {
+    icelake::ondemand::field json_field;
+    SIMDJSON_TRY(std::move(field_result).get(json_field));
+    std::string_view key;
+    SIMDJSON_TRY(json_field.unescaped_key().get(key));
+    std::size_t key_index = keys.size();
+    for (std::size_t i = 0; i < keys.size(); ++i) {
+      if (key == std::string_view(keys[i])) { key_index = i; break; }
+    }
+    if (key_index == keys.size()) {
+      if constexpr (DenyUnknown) {
+        // A key that T itself serializes (e.g. of a const member) is not
+        // unknown: a serialized value must parse back.
+        static constexpr auto ignored_keys = std::define_static_array(unassignable_keys(^^T));
+        bool ignored = false;
+        for (const char *ignored_key : ignored_keys) {
+          if (key == std::string_view(ignored_key)) { ignored = true; break; }
+        }
+        if (!ignored) { return UNKNOWN_FIELD; }
+      }
+      continue;
+    }
+    const std::size_t field_index = key_fields[key_index];
+    if (seen_field[field_index]) { continue; }
+    seen_field[field_index] = true;
+    SIMDJSON_TRY(deserialize_field_at(field_index, json_field.value(), out));
+  }
+  return handle_missing_fields(seen_field, out);
+}
+
+template <typename T>
+consteval bool is_transparent() {
+  return simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag);
+}
+
+} // namespace key_selector_reflection_detail
+
+// Deserialize a reflected struct. By default this builds a compile-time
+// key_selector from the struct's members and walks each object once with
+// object::for_each (perfect-hash key matching), instead of one obj[key] lookup
+// per member. Other paths are used instead:
+//   - globally, the ordered per-member path when defining
+//     -DSIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION=1;
+//   - automatically and per-type, a scan of the object comparing unescaped keys
+//     when the struct's keys do not fit the key_selector limits (see
+//     keys_fit_selector) or cannot be compared raw (see keys_need_unescaping),
+//     so that long member names and the like keep compiling rather than
+//     tripping a static_assert.
+// Structs annotated with deny_unknown_fields always use the strict scan, and
+// structs annotated with transparent are deserialized as their single member.
+//
+// noexcept(false): a member's tag_invoke may throw and the exception must reach
+// the caller of get<T>().
 template <typename T, typename ValT>
   requires(user_defined_type<T> && std::is_class_v<T>)
-error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
+  if constexpr (key_selector_reflection_detail::is_transparent<T>()) {
+    constexpr auto mem = simdjson::detail::transparent_member(^^T);
+    if constexpr (std::is_same_v<std::remove_cvref_t<ValT>, icelake::ondemand::object>) {
+      // We were handed an object: only a structure can be deserialized from it.
+      if constexpr (user_defined_type<typename [: std::meta::type_of(mem) :]>) {
+        return tag_invoke(deserialize_tag{}, val, out.[:mem:]);
+      } else {
+        return INCORRECT_TYPE;
+      }
+    } else {
+      return key_selector_reflection_detail::deserialize_member<mem>(val, out.[:mem:]);
+    }
+  } else {
+  static_assert(key_selector_reflection_detail::accepted_keys_are_distinct<T>(),
+                "two members of this structure accept the same JSON key (check rename, alias, "
+                "rename_all and flatten)");
   icelake::ondemand::object obj;
   if constexpr (std::is_same_v<std::remove_cvref_t<ValT>, icelake::ondemand::object>) {
     obj = val;
   } else {
     SIMDJSON_TRY(val.get_object().get(obj));
   }
-  template for (constexpr auto mem : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
-    if constexpr (!std::meta::is_const(mem) && std::meta::is_public(mem)) {
-      constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
-      if constexpr (concepts::optional_type<decltype(out.[:mem:])>) {
-        // for optional members, it's ok if the key is missing
-        auto error = obj[key].get(out.[:mem:]);
-        if (error && error != NO_SUCH_FIELD) {
-          if(error == NO_SUCH_FIELD) {
-            out.[:mem:].reset();
-            continue;
-          }
-          return error;
-        }
-      } else {
-        // for non-optional members, the key must be present
-        SIMDJSON_TRY(obj[key].get(out.[:mem:]));
+  if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::deny_unknown_fields_tag)) {
+    return key_selector_reflection_detail::deserialize_struct_scan<true>(obj, out);
+  } else {
+#if defined(SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION) && SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION
+  // Opt-out build: use the ordered per-member path, unless obj[key] cannot
+  // match T's keys.
+  if constexpr (key_selector_reflection_detail::keys_need_unescaping<T>()) {
+    return key_selector_reflection_detail::deserialize_struct_scan<false>(obj, out);
+  } else {
+    return key_selector_reflection_detail::deserialize_struct_ordered(obj, out);
+  }
+#else
+  if constexpr (key_selector_reflection_detail::eligible_field_count<T>() == 0) {
+    // No fields to deserialize: an empty key_selector cannot be built, so just
+    // validate that the input is an object (done above) and succeed. Mirrors the
+    // ordered per-member path, which iterates over zero members.
+    (void)out;
+    (void)obj;
+    return SUCCESS;
+  } else if constexpr (!key_selector_reflection_detail::keys_fit_selector<T>()) {
+    // Automatic fallback: T's accepted keys do not fit the key_selector limits
+    // (e.g. a member name longer than 63 characters, or a key with a double
+    // quote), so building a selector would be a compile error. Scan the object
+    // instead, so the default never breaks a struct that the opt-out path would
+    // accept.
+    return key_selector_reflection_detail::deserialize_struct_scan<false>(obj, out);
+  } else {
+  using selector = key_selector_reflection_detail::selector_for<T>;
+  if constexpr (key_selector_reflection_detail::all_eligible_fields_required<T>()
+                && !key_selector_reflection_detail::has_aliases<T>()) {
+    // Fast path: every member is required and has a single key. A single
+    // for_each pass parses each matched field; the returned match count then
+    // tells us whether every member was present (matched_count ==
+    // selector::size()) without a per-member "seen" array. A value-parse error
+    // (e.g. a type mismatch) is propagated by for_each.
+    auto walk = obj.template for_each<selector>(
+        [&](std::size_t matched_index, icelake::ondemand::value field_value) -> error_code {
+      std::size_t counter = 0;
+      template for (constexpr auto path : std::define_static_array(key_selector_reflection_detail::eligible_fields(^^T))) {
+        using field = [: path :];
+        if (matched_index == counter) { return key_selector_reflection_detail::deserialize_member<field::leaf>(field_value, field::get(out)); }
+        ++counter;
       }
-    }
-  };
-  return simdjson::SUCCESS;
-}
-
-// Support for enum deserialization - deserialize from string representation using expand approach from P2996R12
+      return SUCCESS;
+    });
+    if (walk.error) { return walk.error; }
+    // A missing required member shows up as a short match count and is reported as
+    // NO_SUCH_FIELD, mirroring the ordered obj[key] path.
+    if (walk.matched_count != selector::size()) { return NO_SUCH_FIELD; }
+    return SUCCESS;
+  } else {
+    static constexpr auto key_fields = std::define_static_array(key_selector_reflection_detail::accepted_key_fields(^^T));
+    std::array<bool, key_selector_reflection_detail::eligible_field_count<T>()> seen_field{};
+    // Single pass over the object: each field whose key matches a member (or one
+    // of its aliases) yields its selector index, which we map back to the
+    // corresponding member. The first key seen for a member wins. The callback
+    // returns an error_code so that a value-parse error (e.g. a type mismatch on
+    // a matched field) is propagated by for_each instead of being silently dropped.
+    error_code walk_error = obj.template for_each<selector>(
+        [&](std::size_t matched_index, icelake::ondemand::value field_value) -> error_code {
+      const std::size_t field_index = key_fields[matched_index];
+      if (seen_field[field_index]) { return SUCCESS; }
+      seen_field[field_index] = true;
+      return key_selector_reflection_detail::deserialize_field_at(field_index, field_value, out);
+    });
+    if (walk_error) { return walk_error; }
+    // Required members must be present: a missing one is reported as
+    // NO_SUCH_FIELD, mirroring the ordered obj[key] path. Optional and defaulted
+    // members may be absent.
+    return key_selector_reflection_detail::handle_missing_fields(seen_field, out);
+  }
+  }
+#endif // SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION
+  }
+  }
+}
+
+// Support for enum deserialization - deserialize from string representation using
+// expand approach from P2996R12. The accepted strings are the enumerator's JSON key
+// (see rename and rename_all) and its aliases.
 template <typename T, typename ValT>
   requires(std::is_enum_v<T>)
 error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
 #if SIMDJSON_STATIC_REFLECTION
   std::string_view str;
   SIMDJSON_TRY(val.get_string().get(str));
-  constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
+  static constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
   template for (constexpr auto enum_val : enumerators) {
-    if (str == std::meta::identifier_of(enum_val)) {
-      out = [:enum_val:];
-      return SUCCESS;
+    template for (constexpr const char *key : std::define_static_array(key_selector_reflection_detail::accepted_keys_of(enum_val))) {
+      if (str == std::string_view(key)) {
+        out = [:enum_val:];
+        return SUCCESS;
+      }
     }
   };

@@ -109452,33 +139429,25 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {

 template <typename simdjson_value, typename T>
   requires(user_defined_type<std::remove_cvref_t<T>>)
-error_code tag_invoke(deserialize_tag, simdjson_value &val, std::unique_ptr<T> &out) noexcept {
-  if (!out) {
-    out = std::make_unique<T>();
-    if (!out) {
-      return MEMALLOC;
-    }
-  }
-  if (auto err = val.get(*out)) {
-    out.reset();
-    return err;
+error_code tag_invoke(deserialize_tag, simdjson_value &val, std::unique_ptr<T> &out) noexcept(false) {
+  std::unique_ptr<T> ptr(new (std::nothrow) T());
+  if (!ptr) {
+    return MEMALLOC;
   }
+  SIMDJSON_TRY(val.get(*ptr));
+  out = std::move(ptr);
   return SUCCESS;
 }

 template <typename simdjson_value, typename T>
   requires(user_defined_type<std::remove_cvref_t<T>>)
-error_code tag_invoke(deserialize_tag, simdjson_value &val, std::shared_ptr<T> &out) noexcept {
-  if (!out) {
-    out = std::make_shared<T>();
-    if (!out) {
-      return MEMALLOC;
-    }
-  }
-  if (auto err = val.get(*out)) {
-    out.reset();
-    return err;
+error_code tag_invoke(deserialize_tag, simdjson_value &val, std::shared_ptr<T> &out) noexcept(false) {
+  std::shared_ptr<T> ptr(new (std::nothrow) T());
+  if (!ptr) {
+    return MEMALLOC;
   }
+  SIMDJSON_TRY(val.get(*ptr));
+  out = std::move(ptr);
   return SUCCESS;
 }

@@ -109790,9 +139759,17 @@ simdjson_inline simdjson_result<array> array::started(value_iterator &iter) noex
   return array(iter);
 }

-simdjson_inline simdjson_result<array_iterator> array::begin() noexcept {
+simdjson_inline simdjson_result<array_iterator> array::begin() & noexcept {
 #if SIMDJSON_DEVELOPMENT_CHECKS
-  if (!iter.is_at_iterator_start()) { return OUT_OF_ORDER_ITERATION; }
+  if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
+  return array_iterator(iter, this);
+#endif
+  return array_iterator(iter);
+}
+simdjson_inline simdjson_result<array_iterator> array::begin() && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  // The array is a temporary that the iterator may outlive: do not lock it.
+  if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
 #endif
   return array_iterator(iter);
 }
@@ -109819,6 +139796,9 @@ simdjson_inline simdjson_result<std::string_view> array::raw_json() noexcept {
 SIMDJSON_PUSH_DISABLE_WARNINGS
 SIMDJSON_DISABLE_STRICT_OVERFLOW_WARNING
 simdjson_inline simdjson_result<size_t> array::count_elements() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   size_t count{0};
   // Important: we do not consume any of the values.
   for(simdjson_unused auto v : *this) { count++; }
@@ -109832,6 +139812,9 @@ simdjson_inline simdjson_result<size_t> array::count_elements() & noexcept {
 SIMDJSON_POP_DISABLE_WARNINGS

 simdjson_inline simdjson_result<bool> array::is_empty() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   bool is_not_empty;
   auto error = iter.reset_array().get(is_not_empty);
   if(error) { return error; }
@@ -109839,31 +139822,30 @@ simdjson_inline simdjson_result<bool> array::is_empty() & noexcept {
 }

 inline simdjson_result<bool> array::reset() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   return iter.reset_array();
 }

+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline void array::set_locked(bool _locked) noexcept {
+  locked = _locked;
+}
+#endif
+
 inline simdjson_result<value> array::at_pointer(std::string_view json_pointer) noexcept {
-  if (json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
+  // An empty pointer has no json_pointer[0]: with a default-constructed
+  // std::string_view, reading it dereferences a null pointer.
+  if (json_pointer.empty() || json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
   json_pointer = json_pointer.substr(1);
   // - means "the append position" or "the element after the end of the array"
   // We don't support this, because we're returning a real element, not a position.
   if (json_pointer == "-") { return INDEX_OUT_OF_BOUNDS; }

-  // Read the array index
   size_t array_index = 0;
   size_t i;
-  for (i = 0; i < json_pointer.length() && json_pointer[i] != '/'; i++) {
-    uint8_t digit = uint8_t(json_pointer[i] - '0');
-    // Check for non-digit in array index. If it's there, we're trying to get a field in an object
-    if (digit > 9) { return INCORRECT_TYPE; }
-    array_index = array_index*10 + digit;
-  }
-
-  // 0 followed by other digits is invalid
-  if (i > 1 && json_pointer[0] == '0') { return INVALID_JSON_POINTER; } // "JSON pointer array index has other characters after 0"
-
-  // Empty string is invalid; so is a "/" with no digits before it
-  if (i == 0) { return INVALID_JSON_POINTER; } // "Empty string in JSON pointer array index"
+  SIMDJSON_TRY(internal::parse_json_pointer_array_index(json_pointer, array_index, i));
   // Get the child
   auto child = at(array_index);
   // If there is an error, it ends here
@@ -109937,6 +139919,9 @@ inline error_code array::for_each_at_path_with_wildcard(std::string_view json_pa
 }

 simdjson_inline simdjson_result<value> array::at(size_t index) noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   size_t i = 0;
   for (auto value : *this) {
     if (i == index) { return value; }
@@ -109966,10 +139951,14 @@ simdjson_inline simdjson_result<icelake::ondemand::array>::simdjson_result(
 {
 }

-simdjson_inline simdjson_result<icelake::ondemand::array_iterator> simdjson_result<icelake::ondemand::array>::begin() noexcept {
+simdjson_inline simdjson_result<icelake::ondemand::array_iterator> simdjson_result<icelake::ondemand::array>::begin() & noexcept {
   if (error()) { return error(); }
   return first.begin();
 }
+simdjson_inline simdjson_result<icelake::ondemand::array_iterator> simdjson_result<icelake::ondemand::array>::begin() && noexcept {
+  if (error()) { return error(); }
+  return std::move(first).begin();
+}
 simdjson_inline simdjson_result<icelake::ondemand::array_iterator> simdjson_result<icelake::ondemand::array>::end() noexcept {
   if (error()) { return error(); }
   return first.end();
@@ -110032,6 +140021,59 @@ simdjson_inline array_iterator::array_iterator(const value_iterator &_iter) noex
   : iter{_iter}
 {}

+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline array_iterator::array_iterator(const value_iterator &_iter, array* _parent) noexcept
+  : parent{_parent}, iter{_iter}
+{
+  if (parent) parent->set_locked(true);
+}
+
+simdjson_inline array_iterator::~array_iterator() noexcept
+{
+  if (parent) parent->set_locked(false);
+}
+
+simdjson_inline array_iterator::array_iterator(array_iterator&& other) noexcept
+  : has_been_referenced{other.has_been_referenced},
+    parent{other.parent},
+    iter{std::move(other.iter)}
+{
+  other.parent = nullptr;
+}
+
+simdjson_inline array_iterator& array_iterator::operator=(array_iterator&& other) noexcept {
+  if (this != &other)
+  {
+    if (parent)
+      parent->set_locked(false);
+    has_been_referenced = other.has_been_referenced;
+    parent = other.parent;
+    iter = std::move(other.iter);
+
+    other.parent = nullptr;
+  }
+  return *this;
+}
+
+simdjson_inline array_iterator::array_iterator(const array_iterator& other) noexcept
+  : has_been_referenced{other.has_been_referenced},
+    parent{nullptr},
+    iter{other.iter}
+{}
+
+simdjson_inline array_iterator& array_iterator::operator=(const array_iterator& other) noexcept {
+  if (this != &other)
+  {
+    if (parent)
+      parent->set_locked(false);
+    has_been_referenced = other.has_been_referenced;
+    parent = nullptr;
+    iter = other.iter;
+  }
+  return *this;
+}
+#endif
+
 simdjson_inline simdjson_result<value> array_iterator::operator*() noexcept {
 #if SIMDJSON_DEVELOPMENT_CHECKS
    SIMDJSON_ASSUME(!has_been_referenced);
@@ -110127,6 +140169,41 @@ namespace simdjson {
 namespace icelake {
 namespace ondemand {

+/**
+ * @private Narrows the result of get_uint64() to a smaller unsigned integer type T.
+ * @returns NUMBER_OUT_OF_RANGE if the value does not fit in a T.
+ */
+template <typename T>
+simdjson_inline simdjson_result<T> narrow_integer(simdjson_result<uint64_t> wide) noexcept {
+  uint64_t result;
+  SIMDJSON_TRY(std::move(wide).get(result));
+  if (result > static_cast<uint64_t>((std::numeric_limits<T>::max)())) { return NUMBER_OUT_OF_RANGE; }
+  return static_cast<T>(result);
+}
+
+/**
+ * @private Narrows the result of get_int64() to a smaller signed integer type T.
+ * @returns NUMBER_OUT_OF_RANGE if the value does not fit in a T.
+ */
+template <typename T>
+simdjson_inline simdjson_result<T> narrow_integer(simdjson_result<int64_t> wide) noexcept {
+  int64_t result;
+  SIMDJSON_TRY(std::move(wide).get(result));
+  if (result > static_cast<int64_t>((std::numeric_limits<T>::max)()) || result < static_cast<int64_t>((std::numeric_limits<T>::min)())) { return NUMBER_OUT_OF_RANGE; }
+  return static_cast<T>(result);
+}
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+// get_float32() parses with get_float(), so float must be binary32.
+static_assert(std::numeric_limits<float>::is_iec559 && std::numeric_limits<float>::digits == 24,
+              "get_float32() requires float to be IEEE binary32");
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+// get_float64() parses with get_double(), so double must be binary64.
+static_assert(std::numeric_limits<double>::is_iec559 && std::numeric_limits<double>::digits == 53,
+              "get_float64() requires double to be IEEE binary64");
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
+
 simdjson_inline value::value(const value_iterator &_iter) noexcept
   : iter{_iter}
 {
@@ -110158,6 +140235,13 @@ simdjson_inline simdjson_result<raw_json_string> value::get_raw_json_string() no
 simdjson_inline simdjson_result<std::string_view> value::get_string(bool allow_replacement) noexcept {
   return iter.get_string(allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> value::get_u8string(bool allow_replacement) noexcept {
+  std::string_view content;
+  SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+  return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code value::get_string(string_type& receiver, bool allow_replacement) noexcept {
   return iter.get_string(receiver, allow_replacement);
@@ -110171,6 +140255,12 @@ simdjson_inline simdjson_result<double> value::get_double() noexcept {
 simdjson_inline simdjson_result<double> value::get_double_in_string() noexcept {
   return iter.get_double_in_string();
 }
+simdjson_inline simdjson_result<float> value::get_float() noexcept {
+  return iter.get_float();
+}
+simdjson_inline simdjson_result<float> value::get_float_in_string() noexcept {
+  return iter.get_float_in_string();
+}
 simdjson_inline simdjson_result<uint64_t> value::get_uint64() noexcept {
   return iter.get_uint64();
 }
@@ -110184,17 +140274,37 @@ simdjson_inline simdjson_result<int64_t> value::get_int64_in_string() noexcept {
   return iter.get_int64_in_string();
 }
 simdjson_inline simdjson_result<uint32_t> value::get_uint32() noexcept {
-  uint64_t result;
-  SIMDJSON_TRY(get_uint64().get(result));
-  if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<uint32_t>(result);
+  return narrow_integer<uint32_t>(get_uint64());
 }
 simdjson_inline simdjson_result<int32_t> value::get_int32() noexcept {
-  int64_t result;
-  SIMDJSON_TRY(get_int64().get(result));
-  if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<int32_t>(result);
+  return narrow_integer<int32_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint16_t> value::get_uint16() noexcept {
+  return narrow_integer<uint16_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int16_t> value::get_int16() noexcept {
+  return narrow_integer<int16_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint8_t> value::get_uint8() noexcept {
+  return narrow_integer<uint8_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int8_t> value::get_int8() noexcept {
+  return narrow_integer<int8_t>(get_int64());
 }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> value::get_float32() noexcept {
+  float result;
+  SIMDJSON_TRY(get_float().get(result));
+  return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> value::get_float64() noexcept {
+  double result;
+  SIMDJSON_TRY(get_double().get(result));
+  return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<bool> value::get_bool() noexcept {
   return iter.get_bool();
 }
@@ -110206,12 +140316,26 @@ template<> simdjson_inline simdjson_result<array> value::get() noexcept { return
 template<> simdjson_inline simdjson_result<object> value::get() noexcept { return get_object(); }
 template<> simdjson_inline simdjson_result<raw_json_string> value::get() noexcept { return get_raw_json_string(); }
 template<> simdjson_inline simdjson_result<std::string_view> value::get() noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> value::get() noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_inline simdjson_result<number> value::get() noexcept { return get_number(); }
 template<> simdjson_inline simdjson_result<double> value::get() noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> value::get() noexcept { return get_float(); }
 template<> simdjson_inline simdjson_result<uint64_t> value::get() noexcept { return get_uint64(); }
 template<> simdjson_inline simdjson_result<int64_t> value::get() noexcept { return get_int64(); }
 template<> simdjson_inline simdjson_result<uint32_t> value::get() noexcept { return get_uint32(); }
 template<> simdjson_inline simdjson_result<int32_t> value::get() noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> value::get() noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> value::get() noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> value::get() noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> value::get() noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> value::get() noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> value::get() noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<> simdjson_inline simdjson_result<bool> value::get() noexcept { return get_bool(); }


@@ -110219,12 +140343,26 @@ template<> simdjson_warn_unused simdjson_inline error_code value::get(array& out
 template<> simdjson_warn_unused simdjson_inline error_code value::get(object& out) noexcept { return get_object().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(raw_json_string& out) noexcept { return get_raw_json_string().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(std::string_view& out) noexcept { return get_string(false).get(out); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::u8string_view& out) noexcept { return get_u8string(false).get(out); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_warn_unused simdjson_inline error_code value::get(number& out) noexcept { return get_number().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(double& out) noexcept { return get_double().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(float& out) noexcept { return get_float().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(uint64_t& out) noexcept { return get_uint64().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(int64_t& out) noexcept { return get_int64().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(uint32_t& out) noexcept { return get_uint32().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(int32_t& out) noexcept { return get_int32().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(uint16_t& out) noexcept { return get_uint16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(int16_t& out) noexcept { return get_int16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(uint8_t& out) noexcept { return get_uint8().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(int8_t& out) noexcept { return get_int8().get(out); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::float32_t& out) noexcept { return get_float32().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::float64_t& out) noexcept { return get_float64().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<>  simdjson_warn_unused simdjson_inline error_code value::get(bool& out) noexcept { return get_bool().get(out); }

 #if SIMDJSON_EXCEPTIONS
@@ -110393,6 +140531,9 @@ inline bool is_pointer_well_formed(std::string_view json_pointer) noexcept {
 }

 simdjson_inline simdjson_result<value> value::at_pointer(std::string_view json_pointer) noexcept {
+  // The empty JSON Pointer refers to the whole value (RFC 6901), as in
+  // document::at_pointer.
+  if (json_pointer.empty()) { return value(iter); }
   json_type t;
   SIMDJSON_TRY(type().get(t));
   switch (t)
@@ -110430,6 +140571,10 @@ template <typename Func>
 template <typename Func>
 #endif
 inline error_code value::for_each_at_path_with_wildcard(std::string_view json_path, Func&& callback) noexcept {
+  // Every recursive step of for_each_at_path_with_wildcard goes through this
+  // function, and each one descends one level into the document. A path with
+  // many segments applied to a deeply nested document would otherwise recurse
+  // without bound and overflow the stack.
   if (size_t(iter.depth()) >= iter.json_iter().parser->max_depth()) { return DEPTH_ERROR; }
   json_type t;
   SIMDJSON_TRY(type().get(t));
@@ -110543,10 +140688,46 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<icelake::ondemand::valu
   if (error()) { return error(); }
   return first.get_int32();
 }
+simdjson_inline simdjson_result<uint16_t> simdjson_result<icelake::ondemand::value>::get_uint16() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<icelake::ondemand::value>::get_int16() noexcept {
+  if (error()) { return error(); }
+  return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<icelake::ondemand::value>::get_uint8() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<icelake::ondemand::value>::get_int8() noexcept {
+  if (error()) { return error(); }
+  return first.get_int8();
+}
 simdjson_inline simdjson_result<double> simdjson_result<icelake::ondemand::value>::get_double() noexcept {
   if (error()) { return error(); }
   return first.get_double();
 }
+simdjson_inline simdjson_result<float> simdjson_result<icelake::ondemand::value>::get_float() noexcept {
+  if (error()) { return error(); }
+  return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<icelake::ondemand::value>::get_float_in_string() noexcept {
+  if (error()) { return error(); }
+  return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<icelake::ondemand::value>::get_float32() noexcept {
+  if (error()) { return error(); }
+  return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<icelake::ondemand::value>::get_float64() noexcept {
+  if (error()) { return error(); }
+  return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<double> simdjson_result<icelake::ondemand::value>::get_double_in_string() noexcept {
   if (error()) { return error(); }
   return first.get_double_in_string();
@@ -110555,6 +140736,12 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<icelake::ondem
   if (error()) { return error(); }
   return first.get_string(allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<icelake::ondemand::value>::get_u8string(bool allow_replacement) noexcept {
+  if (error()) { return error(); }
+  return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_inline error_code simdjson_result<icelake::ondemand::value>::get_string(string_type& receiver, bool allow_replacement) noexcept {
   if (error()) { return error(); }
@@ -110583,11 +140770,23 @@ template<> simdjson_inline error_code simdjson_result<icelake::ondemand::value>:
   return SUCCESS;
 }

-template<typename T> simdjson_inline simdjson_result<T> simdjson_result<icelake::ondemand::value>::get() noexcept {
+template<typename T> simdjson_inline simdjson_result<T> simdjson_result<icelake::ondemand::value>::get()
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, icelake::ondemand::value>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>();
 }
-template<typename T> simdjson_inline error_code simdjson_result<icelake::ondemand::value>::get(T &out) noexcept {
+template<typename T> simdjson_inline error_code simdjson_result<icelake::ondemand::value>::get(T &out)
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, icelake::ondemand::value>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>(out);
 }
@@ -110857,16 +141056,22 @@ simdjson_inline simdjson_result<int64_t> document::get_int64_in_string() noexcep
   return get_root_value_iterator().get_root_int64_in_string(true);
 }
 simdjson_inline simdjson_result<uint32_t> document::get_uint32() noexcept {
-  uint64_t result;
-  SIMDJSON_TRY(get_uint64().get(result));
-  if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<uint32_t>(result);
+  return narrow_integer<uint32_t>(get_uint64());
 }
 simdjson_inline simdjson_result<int32_t> document::get_int32() noexcept {
-  int64_t result;
-  SIMDJSON_TRY(get_int64().get(result));
-  if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<int32_t>(result);
+  return narrow_integer<int32_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint16_t> document::get_uint16() noexcept {
+  return narrow_integer<uint16_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int16_t> document::get_int16() noexcept {
+  return narrow_integer<int16_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint8_t> document::get_uint8() noexcept {
+  return narrow_integer<uint8_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int8_t> document::get_int8() noexcept {
+  return narrow_integer<int8_t>(get_int64());
 }
 simdjson_inline simdjson_result<double> document::get_double() noexcept {
   return get_root_value_iterator().get_root_double(true);
@@ -110874,9 +141079,36 @@ simdjson_inline simdjson_result<double> document::get_double() noexcept {
 simdjson_inline simdjson_result<double> document::get_double_in_string() noexcept {
   return get_root_value_iterator().get_root_double_in_string(true);
 }
+simdjson_inline simdjson_result<float> document::get_float() noexcept {
+  return get_root_value_iterator().get_root_float(true);
+}
+simdjson_inline simdjson_result<float> document::get_float_in_string() noexcept {
+  return get_root_value_iterator().get_root_float_in_string(true);
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> document::get_float32() noexcept {
+  float result;
+  SIMDJSON_TRY(get_float().get(result));
+  return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> document::get_float64() noexcept {
+  double result;
+  SIMDJSON_TRY(get_double().get(result));
+  return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<std::string_view> document::get_string(bool allow_replacement) noexcept {
   return get_root_value_iterator().get_root_string(true, allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> document::get_u8string(bool allow_replacement) noexcept {
+  std::string_view content;
+  SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+  return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code document::get_string(string_type& receiver, bool allow_replacement) noexcept {
   return get_root_value_iterator().get_root_string(receiver, true, allow_replacement);
@@ -110898,11 +141130,25 @@ template<> simdjson_inline simdjson_result<array> document::get() & noexcept { r
 template<> simdjson_inline simdjson_result<object> document::get() & noexcept { return get_object(); }
 template<> simdjson_inline simdjson_result<raw_json_string> document::get() & noexcept { return get_raw_json_string(); }
 template<> simdjson_inline simdjson_result<std::string_view> document::get() & noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> document::get() & noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_inline simdjson_result<double> document::get() & noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> document::get() & noexcept { return get_float(); }
 template<> simdjson_inline simdjson_result<uint64_t> document::get() & noexcept { return get_uint64(); }
 template<> simdjson_inline simdjson_result<int64_t> document::get() & noexcept { return get_int64(); }
 template<> simdjson_inline simdjson_result<uint32_t> document::get() & noexcept { return get_uint32(); }
 template<> simdjson_inline simdjson_result<int32_t> document::get() & noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> document::get() & noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> document::get() & noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> document::get() & noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> document::get() & noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> document::get() & noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> document::get() & noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<> simdjson_inline simdjson_result<bool> document::get() & noexcept { return get_bool(); }
 template<> simdjson_inline simdjson_result<value> document::get() & noexcept { return get_value(); }

@@ -110910,17 +141156,35 @@ template<> simdjson_warn_unused simdjson_inline error_code document::get(array&
 template<> simdjson_warn_unused simdjson_inline error_code document::get(object& out) & noexcept { return get_object().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(raw_json_string& out) & noexcept { return get_raw_json_string().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(std::string_view& out) & noexcept { return get_string(false).get(out); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::u8string_view& out) & noexcept { return get_u8string(false).get(out); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_warn_unused simdjson_inline error_code document::get(double& out) & noexcept { return get_double().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(float& out) & noexcept { return get_float().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(uint64_t& out) & noexcept { return get_uint64().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(int64_t& out) & noexcept { return get_int64().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(uint32_t& out) & noexcept { return get_uint32().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(int32_t& out) & noexcept { return get_int32().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(uint16_t& out) & noexcept { return get_uint16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(int16_t& out) & noexcept { return get_int16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(uint8_t& out) & noexcept { return get_uint8().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(int8_t& out) & noexcept { return get_int8().get(out); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::float32_t& out) & noexcept { return get_float32().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::float64_t& out) & noexcept { return get_float64().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<> simdjson_warn_unused simdjson_inline error_code document::get(bool& out) & noexcept { return get_bool().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(value& out) & noexcept { return get_value().get(out); }

 template<> simdjson_deprecated simdjson_inline simdjson_result<raw_json_string> document::get() && noexcept { return get_raw_json_string(); }
 template<> simdjson_deprecated simdjson_inline simdjson_result<std::string_view> document::get() && noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_deprecated simdjson_inline simdjson_result<std::u8string_view> document::get() && noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_deprecated simdjson_inline simdjson_result<double> document::get() && noexcept { return std::forward<document>(*this).get_double(); }
+template<> simdjson_deprecated simdjson_inline simdjson_result<float> document::get() && noexcept { return std::forward<document>(*this).get_float(); }
 template<> simdjson_deprecated simdjson_inline simdjson_result<uint64_t> document::get() && noexcept { return std::forward<document>(*this).get_uint64(); }
 template<> simdjson_deprecated simdjson_inline simdjson_result<int64_t> document::get() && noexcept { return std::forward<document>(*this).get_int64(); }
 template<> simdjson_deprecated simdjson_inline simdjson_result<bool> document::get() && noexcept { return std::forward<document>(*this).get_bool(); }
@@ -111259,6 +141523,22 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<icelake::ondemand::docu
   if (error()) { return error(); }
   return first.get_int32();
 }
+simdjson_inline simdjson_result<uint16_t> simdjson_result<icelake::ondemand::document>::get_uint16() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<icelake::ondemand::document>::get_int16() noexcept {
+  if (error()) { return error(); }
+  return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<icelake::ondemand::document>::get_uint8() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<icelake::ondemand::document>::get_int8() noexcept {
+  if (error()) { return error(); }
+  return first.get_int8();
+}
 simdjson_inline simdjson_result<double> simdjson_result<icelake::ondemand::document>::get_double() noexcept {
   if (error()) { return error(); }
   return first.get_double();
@@ -111267,10 +141547,36 @@ simdjson_inline simdjson_result<double> simdjson_result<icelake::ondemand::docum
   if (error()) { return error(); }
   return first.get_double_in_string();
 }
+simdjson_inline simdjson_result<float> simdjson_result<icelake::ondemand::document>::get_float() noexcept {
+  if (error()) { return error(); }
+  return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<icelake::ondemand::document>::get_float_in_string() noexcept {
+  if (error()) { return error(); }
+  return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<icelake::ondemand::document>::get_float32() noexcept {
+  if (error()) { return error(); }
+  return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<icelake::ondemand::document>::get_float64() noexcept {
+  if (error()) { return error(); }
+  return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<std::string_view> simdjson_result<icelake::ondemand::document>::get_string(bool allow_replacement) noexcept {
   if (error()) { return error(); }
   return first.get_string(allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<icelake::ondemand::document>::get_u8string(bool allow_replacement) noexcept {
+  if (error()) { return error(); }
+  return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code simdjson_result<icelake::ondemand::document>::get_string(string_type& receiver, bool allow_replacement) noexcept {
   if (error()) { return error(); }
@@ -111298,22 +141604,46 @@ simdjson_inline simdjson_result<bool> simdjson_result<icelake::ondemand::documen
 }

 template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<icelake::ondemand::document>::get() & noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<icelake::ondemand::document>::get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, icelake::ondemand::document>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>();
 }
 template<typename T>
-simdjson_deprecated simdjson_inline simdjson_result<T> simdjson_result<icelake::ondemand::document>::get() && noexcept {
+simdjson_deprecated simdjson_inline simdjson_result<T> simdjson_result<icelake::ondemand::document>::get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, icelake::ondemand::document>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return std::forward<icelake::ondemand::document>(first).get<T>();
 }
 template<typename T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<icelake::ondemand::document>::get(T &out) & noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<icelake::ondemand::document>::get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, icelake::ondemand::document>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>(out);
 }
 template<typename T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<icelake::ondemand::document>::get(T &out) && noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<icelake::ondemand::document>::get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, icelake::ondemand::document>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return std::forward<icelake::ondemand::document>(first).get<T>(out);
 }
@@ -111382,27 +141712,27 @@ simdjson_inline simdjson_result<icelake::ondemand::document>::operator icelake::
 }
 simdjson_inline simdjson_result<icelake::ondemand::document>::operator uint64_t() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_uint64();
 }
 simdjson_inline simdjson_result<icelake::ondemand::document>::operator int64_t() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_int64();
 }
 simdjson_inline simdjson_result<icelake::ondemand::document>::operator double() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_double();
 }
 simdjson_inline simdjson_result<icelake::ondemand::document>::operator std::string_view() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_string();
 }
 simdjson_inline simdjson_result<icelake::ondemand::document>::operator icelake::ondemand::raw_json_string() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_raw_json_string();
 }
 simdjson_inline simdjson_result<icelake::ondemand::document>::operator bool() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_bool();
 }
 simdjson_inline simdjson_result<icelake::ondemand::document>::operator icelake::ondemand::value() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
@@ -111492,21 +141822,38 @@ simdjson_inline simdjson_result<uint64_t> document_reference::get_uint64() noexc
 simdjson_inline simdjson_result<uint64_t> document_reference::get_uint64_in_string() noexcept { return doc->get_root_value_iterator().get_root_uint64_in_string(false); }
 simdjson_inline simdjson_result<int64_t> document_reference::get_int64() noexcept { return doc->get_root_value_iterator().get_root_int64(false); }
 simdjson_inline simdjson_result<int64_t> document_reference::get_int64_in_string() noexcept { return doc->get_root_value_iterator().get_root_int64_in_string(false); }
-simdjson_inline simdjson_result<uint32_t> document_reference::get_uint32() noexcept {
-  uint64_t result;
-  SIMDJSON_TRY(doc->get_root_value_iterator().get_root_uint64(false).get(result));
-  if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<uint32_t>(result);
-}
-simdjson_inline simdjson_result<int32_t> document_reference::get_int32() noexcept {
-  int64_t result;
-  SIMDJSON_TRY(doc->get_root_value_iterator().get_root_int64(false).get(result));
-  if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<int32_t>(result);
-}
+simdjson_inline simdjson_result<uint32_t> document_reference::get_uint32() noexcept { return narrow_integer<uint32_t>(get_uint64()); }
+simdjson_inline simdjson_result<int32_t> document_reference::get_int32() noexcept { return narrow_integer<int32_t>(get_int64()); }
+simdjson_inline simdjson_result<uint16_t> document_reference::get_uint16() noexcept { return narrow_integer<uint16_t>(get_uint64()); }
+simdjson_inline simdjson_result<int16_t> document_reference::get_int16() noexcept { return narrow_integer<int16_t>(get_int64()); }
+simdjson_inline simdjson_result<uint8_t> document_reference::get_uint8() noexcept { return narrow_integer<uint8_t>(get_uint64()); }
+simdjson_inline simdjson_result<int8_t> document_reference::get_int8() noexcept { return narrow_integer<int8_t>(get_int64()); }
 simdjson_inline simdjson_result<double> document_reference::get_double() noexcept { return doc->get_root_value_iterator().get_root_double(false); }
-simdjson_inline simdjson_result<double> document_reference::get_double_in_string() noexcept { return doc->get_root_value_iterator().get_root_double(false); }
+simdjson_inline simdjson_result<double> document_reference::get_double_in_string() noexcept { return doc->get_root_value_iterator().get_root_double_in_string(false); }
+simdjson_inline simdjson_result<float> document_reference::get_float() noexcept { return doc->get_root_value_iterator().get_root_float(false); }
+simdjson_inline simdjson_result<float> document_reference::get_float_in_string() noexcept { return doc->get_root_value_iterator().get_root_float_in_string(false); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> document_reference::get_float32() noexcept {
+  float result;
+  SIMDJSON_TRY(get_float().get(result));
+  return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> document_reference::get_float64() noexcept {
+  double result;
+  SIMDJSON_TRY(get_double().get(result));
+  return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<std::string_view> document_reference::get_string(bool allow_replacement) noexcept { return doc->get_root_value_iterator().get_root_string(false, allow_replacement); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> document_reference::get_u8string(bool allow_replacement) noexcept {
+  std::string_view content;
+  SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+  return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code document_reference::get_string(string_type& receiver, bool allow_replacement) noexcept { return doc->get_root_value_iterator().get_root_string(receiver, false, allow_replacement); }
 simdjson_inline simdjson_result<std::string_view> document_reference::get_wobbly_string() noexcept { return doc->get_root_value_iterator().get_root_wobbly_string(false); }
@@ -111518,11 +141865,25 @@ template<> simdjson_inline simdjson_result<array> document_reference::get() & no
 template<> simdjson_inline simdjson_result<object> document_reference::get() & noexcept { return get_object(); }
 template<> simdjson_inline simdjson_result<raw_json_string> document_reference::get() & noexcept { return get_raw_json_string(); }
 template<> simdjson_inline simdjson_result<std::string_view> document_reference::get() & noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> document_reference::get() & noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_inline simdjson_result<double> document_reference::get() & noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> document_reference::get() & noexcept { return get_float(); }
 template<> simdjson_inline simdjson_result<uint64_t> document_reference::get() & noexcept { return get_uint64(); }
 template<> simdjson_inline simdjson_result<int64_t> document_reference::get() & noexcept { return get_int64(); }
 template<> simdjson_inline simdjson_result<uint32_t> document_reference::get() & noexcept { return get_uint32(); }
 template<> simdjson_inline simdjson_result<int32_t> document_reference::get() & noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> document_reference::get() & noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> document_reference::get() & noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> document_reference::get() & noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> document_reference::get() & noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> document_reference::get() & noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> document_reference::get() & noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<> simdjson_inline simdjson_result<bool> document_reference::get() & noexcept { return get_bool(); }
 template<> simdjson_inline simdjson_result<value> document_reference::get() & noexcept { return get_value(); }
 #if SIMDJSON_EXCEPTIONS
@@ -111668,6 +142029,22 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<icelake::ondemand::docu
   if (error()) { return error(); }
   return first.get_int32();
 }
+simdjson_inline simdjson_result<uint16_t> simdjson_result<icelake::ondemand::document_reference>::get_uint16() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<icelake::ondemand::document_reference>::get_int16() noexcept {
+  if (error()) { return error(); }
+  return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<icelake::ondemand::document_reference>::get_uint8() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<icelake::ondemand::document_reference>::get_int8() noexcept {
+  if (error()) { return error(); }
+  return first.get_int8();
+}
 simdjson_inline simdjson_result<double> simdjson_result<icelake::ondemand::document_reference>::get_double() noexcept {
   if (error()) { return error(); }
   return first.get_double();
@@ -111676,10 +142053,36 @@ simdjson_inline simdjson_result<double> simdjson_result<icelake::ondemand::docum
   if (error()) { return error(); }
   return first.get_double_in_string();
 }
+simdjson_inline simdjson_result<float> simdjson_result<icelake::ondemand::document_reference>::get_float() noexcept {
+  if (error()) { return error(); }
+  return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<icelake::ondemand::document_reference>::get_float_in_string() noexcept {
+  if (error()) { return error(); }
+  return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<icelake::ondemand::document_reference>::get_float32() noexcept {
+  if (error()) { return error(); }
+  return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<icelake::ondemand::document_reference>::get_float64() noexcept {
+  if (error()) { return error(); }
+  return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<std::string_view> simdjson_result<icelake::ondemand::document_reference>::get_string(bool allow_replacement) noexcept {
   if (error()) { return error(); }
   return first.get_string(allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<icelake::ondemand::document_reference>::get_u8string(bool allow_replacement) noexcept {
+  if (error()) { return error(); }
+  return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code simdjson_result<icelake::ondemand::document_reference>::get_string(string_type& receiver, bool allow_replacement) noexcept {
   if (error()) { return error(); }
@@ -111706,22 +142109,46 @@ simdjson_inline simdjson_result<bool> simdjson_result<icelake::ondemand::documen
   return first.is_null();
 }
 template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<icelake::ondemand::document_reference>::get() & noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<icelake::ondemand::document_reference>::get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, icelake::ondemand::document_reference>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>();
 }
 template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<icelake::ondemand::document_reference>::get() && noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<icelake::ondemand::document_reference>::get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, icelake::ondemand::document_reference>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return std::forward<icelake::ondemand::document_reference>(first).get<T>();
 }
 template <class T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<icelake::ondemand::document_reference>::get(T &out) & noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<icelake::ondemand::document_reference>::get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, icelake::ondemand::document_reference>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>(out);
 }
 template <class T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<icelake::ondemand::document_reference>::get(T &out) && noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<icelake::ondemand::document_reference>::get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, icelake::ondemand::document_reference>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return std::forward<icelake::ondemand::document_reference>(first).get<T>(out);
 }
@@ -111783,27 +142210,27 @@ simdjson_inline simdjson_result<icelake::ondemand::document_reference>::operator
 }
 simdjson_inline simdjson_result<icelake::ondemand::document_reference>::operator uint64_t() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_uint64();
 }
 simdjson_inline simdjson_result<icelake::ondemand::document_reference>::operator int64_t() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_int64();
 }
 simdjson_inline simdjson_result<icelake::ondemand::document_reference>::operator double() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_double();
 }
 simdjson_inline simdjson_result<icelake::ondemand::document_reference>::operator std::string_view() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_string();
 }
 simdjson_inline simdjson_result<icelake::ondemand::document_reference>::operator icelake::ondemand::raw_json_string() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_raw_json_string();
 }
 simdjson_inline simdjson_result<icelake::ondemand::document_reference>::operator bool() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_bool();
 }
 simdjson_inline simdjson_result<icelake::ondemand::document_reference>::operator icelake::ondemand::value() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
@@ -111869,6 +142296,7 @@ simdjson_warn_unused simdjson_inline error_code simdjson_result<icelake::ondeman
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */

 #include <algorithm>
+#include <cstring>
 #include <stdexcept>

 namespace simdjson {
@@ -111955,23 +142383,20 @@ simdjson_inline document_stream::document_stream(
   const uint8_t *_buf,
   size_t _len,
   size_t _batch_size,
-  bool _allow_comma_separated
+  bool _allow_comma_separated,
+  stream_format _format
 ) noexcept
   : parser{&_parser},
     buf{_buf},
     len{_len},
     batch_size{_batch_size <= MINIMAL_BATCH_SIZE ? MINIMAL_BATCH_SIZE : _batch_size},
     allow_comma_separated{_allow_comma_separated},
+    format{_format},
     error{SUCCESS}
     #ifdef SIMDJSON_THREADS_ENABLED
     , use_thread(_parser.threaded) // we need to make a copy because _parser.threaded can change
     #endif
 {
-#ifdef SIMDJSON_THREADS_ENABLED
-  if(worker.get() == nullptr) {
-    error = MEMALLOC;
-  }
-#endif
 }

 simdjson_inline document_stream::document_stream() noexcept
@@ -111980,6 +142405,7 @@ simdjson_inline document_stream::document_stream() noexcept
     len{0},
     batch_size{0},
     allow_comma_separated{false},
+    format{stream_format::whitespace_delimited},
     error{UNINITIALIZED}
     #ifdef SIMDJSON_THREADS_ENABLED
     , use_thread(false)
@@ -111999,6 +142425,9 @@ inline size_t document_stream::size_in_bytes() const noexcept {
 }

 inline size_t document_stream::truncated_bytes() const noexcept {
+  // Stage 1 returns EMPTY on zero-length input before it writes the index
+  // sentinels read below, so they would still hold a previous stream's values.
+  if (len == 0) { return 0; }
   if(error == CAPACITY) { return len - batch_start; }
   return parser->implementation->structural_indexes[parser->implementation->n_structural_indexes] - parser->implementation->structural_indexes[parser->implementation->n_structural_indexes + 1];
 }
@@ -112079,13 +142508,20 @@ inline void document_stream::start() noexcept {
     error = run_stage1(*parser, batch_start);
   }
   if (error) { return; }
-  doc_index = batch_start;
+  // For json_sequence mode, structural_indexes[0] points to the actual JSON value
+  // after the RS delimiter and any following whitespace. For regular mode, it is
+  // the offset from batch_start to the first document in the batch.
+  doc_index = batch_start + parser->implementation->structural_indexes[0];
   doc = document(json_iterator(&buf[batch_start], parser));
   doc.iter._streaming = true;

   #ifdef SIMDJSON_THREADS_ENABLED
   if (use_thread && next_batch_start() < len) {
     // Kick off the first thread on next batch if needed
+    if (worker.get() == nullptr) {
+      worker.reset(new(std::nothrow) stage1_worker());
+      if (worker.get() == nullptr) { error = MEMALLOC; return; }
+    }
     error = stage1_thread_parser.allocate(batch_size);
     if (error) { return; }
     worker->start_thread();
@@ -112160,12 +142596,69 @@ inline void document_stream::next() noexcept {
        */

       if (error) { continue; } // If the error was EMPTY, we may want to load another batch.
-      doc_index = batch_start;
+      doc_index = batch_start + parser->implementation->structural_indexes[0];
     }
   }
 }

+simdjson_inline uint8_t document_stream::document_delimiter() const noexcept {
+  switch (format) {
+    case stream_format::newline_delimited: return '\n';
+    case stream_format::json_sequence: return 0x1E;
+    default: return 0;
+  }
+}
+
+simdjson_inline bool document_stream::skip_to_delimiter(uint8_t delimiter) noexcept {
+  const uint8_t *const base = &buf[batch_start];
+  const token_position pos = doc.iter.position();
+  const token_position end = doc.iter.end_position();
+  if (pos >= end) { return false; }
+  const size_t here = size_t(doc.iter.token.peek(pos) - base);
+  const size_t batch_len =
+      (len - batch_start < batch_size) ? len - batch_start : batch_size;
+  if (here >= batch_len) { return false; }
+  const uint8_t *const found = static_cast<const uint8_t *>(
+      std::memchr(base + here, delimiter, batch_len - here));
+  if (found == nullptr) { return false; }
+
+  const uint32_t boundary = uint32_t(found - base);
+  // The answer is near `pos`: the delimiter ends the current document, while
+  // `end` spans the whole batch. Gallop first so the cost follows the distance
+  // rather than the size of the batch.
+  token_position lo = pos;
+  size_t hop = 1;
+  while (lo + hop < end && lo[hop] < boundary) { lo += hop; hop <<= 1; }
+  token_position hi = (lo + hop < end) ? lo + hop : end;
+  while (lo < hi) {
+    const token_position mid = lo + ((hi - lo) >> 1);
+    if (*mid < boundary) { lo = mid + 1; } else { hi = mid; }
+  }
+  doc.iter.token.set_position(lo);
+  return true;
+}
+
 inline void document_stream::next_document() noexcept {
+  // A delimiter that cannot occur inside a document tells us where the current
+  // one ends, so we can jump there instead of walking every structural. Only
+  // valid while the iterator is still inside the document: a consumed document
+  // already sits on the next one's first token, and skip_child() returns at
+  // once for it.
+  //
+  // The jump does not structure-validate the unread remainder of the current
+  // document: under newline_delimited / json_sequence the next delimiter is
+  // assumed to be the true document boundary. Callers that leave depth() > 0
+  // while violating that contract (e.g. pretty multi-line JSON under
+  // newline_delimited) can mis-align following documents; use
+  // whitespace_delimited if unsure.
+  const uint8_t delimiter = document_delimiter();
+  if (delimiter != 0 && !error && doc.iter.depth() > 0 &&
+      skip_to_delimiter(delimiter)) {
+    doc.iter._depth = 1;
+    doc.iter._string_buf_loc = parser->string_buf.get();
+    doc.iter._root = doc.iter.position();
+    return;
+  }
   // Go to next place where depth=0 (document depth)
   error = doc.iter.skip_child(0);
   if (error) { return; }
@@ -112189,10 +142682,35 @@ inline error_code document_stream::run_stage1(ondemand::parser &p, size_t _batch
   // This code only updates the structural index in the parser, it does not update any json_iterator
   // instance.
   size_t remaining = len - _batch_start;
+  stage1_mode mode;
   if (remaining <= batch_size) {
-    return p.implementation->stage1(&buf[_batch_start], remaining, stage1_mode::streaming_final);
+    // Final batch
+    switch (format) {
+      case stream_format::json_sequence:
+        mode = stage1_mode::json_sequence_final;
+        break;
+      case stream_format::comma_delimited:
+        mode = stage1_mode::comma_delimited_final;
+        break;
+      default:
+        mode = stage1_mode::streaming_final;
+        break;
+    }
+    return p.implementation->stage1(&buf[_batch_start], remaining, mode);
   } else {
-    return p.implementation->stage1(&buf[_batch_start], batch_size, stage1_mode::streaming_partial);
+    // Partial batch
+    switch (format) {
+      case stream_format::json_sequence:
+        mode = stage1_mode::json_sequence_partial;
+        break;
+      case stream_format::comma_delimited:
+        mode = stage1_mode::comma_delimited_partial;
+        break;
+      default:
+        mode = stage1_mode::streaming_partial;
+        break;
+    }
+    return p.implementation->stage1(&buf[_batch_start], batch_size, mode);
   }
 }

@@ -112201,11 +142719,19 @@ simdjson_inline size_t document_stream::iterator::current_index() const noexcept
 }

 simdjson_inline std::string_view document_stream::iterator::source() const noexcept {
-  auto depth = stream->doc.iter.depth();
+  // On error (e.g., CAPACITY), there is no document to walk: return the rest of
+  // the input, as the DOM document_stream does.
+  if (stream->error) {
+    return std::string_view(reinterpret_cast<const char*>(stream->buf) + current_index(), stream->len - current_index());
+  }
+  // Always walk from the root of the document, whatever the current position
+  // of the document iterator: the user may have already consumed part of the
+  // document, so the iterator's current depth must not be used here.
+  depth_t depth = 1;
   auto cur_struct_index = stream->doc.iter._root - stream->parser->implementation->structural_indexes.get();

-  // If at root, process the first token to determine if scalar value
-  if (stream->doc.iter.at_root()) {
+  // Process the first token to determine if scalar value
+  {
     switch (stream->buf[stream->batch_start + stream->parser->implementation->structural_indexes[cur_struct_index]]) {
       case '{': case '[':   // Depth=1 already at start of document
         break;
@@ -112213,14 +142739,47 @@ simdjson_inline std::string_view document_stream::iterator::source() const noexc
         depth--;
         break;
       default:    // Scalar value document
-        // TODO: We could remove trailing whitespaces
         // This returns a string spanning from start of value to the beginning of the next document (excluded)
         {
           auto next_index = stream->batch_start + stream->parser->implementation->structural_indexes[++cur_struct_index];
           // normally the length would be next_index - current_index() - 1, except for the last document
           size_t svlen = next_index - current_index();
           const char *start = reinterpret_cast<const char*>(stream->buf) + current_index();
-          while(svlen > 1 && (std::isspace(start[svlen-1]) || start[svlen-1] == '\0')) {
+          // When the scalar is followed by a truncated document, the structural
+          // indexes of that document were dropped and next_index is the end of
+          // the input, so we bound the scalar by scanning the token itself.
+          size_t token_len = 0;
+          if (*start == '"') {
+            token_len = 1;
+            while (token_len < svlen) {
+              char c = start[token_len++];
+              if (c == '\\') {
+                token_len++;
+              } else if (c == '"') {
+                break;
+              }
+            }
+          } else {
+            while (token_len < svlen) {
+              char c = start[token_len];
+              if (std::isspace(static_cast<unsigned char>(c)) || c == ',' || c == '{' || c == '[' || c == '\0' || static_cast<uint8_t>(c) == 0x1E) {
+                break;
+              }
+              token_len++;
+            }
+          }
+          if (token_len > 0 && token_len < svlen) {
+            svlen = token_len;
+          }
+          // Trim trailing whitespace, NUL, and RS (0x1E). In RFC 7464
+          // json_sequence mode the scanner classifies RS as a scalar
+          // character, so an RS-prefixed scalar document (number / true /
+          // false / null / string) has no closing structural index and the
+          // slice runs all the way up to the next document's RS. RS cannot
+          // legally appear in a JSON value at the source level (control
+          // characters in strings must be escaped as \u001E), so stripping
+          // it is safe in every stream_format.
+          while(svlen > 1 && (std::isspace(static_cast<unsigned char>(start[svlen-1])) || start[svlen-1] == '\0' || static_cast<uint8_t>(start[svlen-1]) == 0x1E || (stream->format == stream_format::comma_delimited && start[svlen-1] == ','))) {
             svlen--;
           }
           return std::string_view(start, svlen);
@@ -112345,11 +142904,19 @@ simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> field::un
   return answer;
 }

+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_warn_unused simdjson_result<std::u8string_view> field::unescaped_u8key(bool allow_replacement) noexcept {
+  std::string_view key;
+  SIMDJSON_TRY( unescaped_key(allow_replacement).get(key) );
+  return internal::as_u8string_view(key);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 template <typename string_type>
 simdjson_inline simdjson_warn_unused error_code field::unescaped_key(string_type& receiver, bool allow_replacement) noexcept {
   std::string_view key;
   SIMDJSON_TRY( unescaped_key(allow_replacement).get(key) );
-  receiver = key;
+  internal::assign_utf8(receiver, key);
   return SUCCESS;
 }

@@ -112371,6 +142938,12 @@ simdjson_inline std::string_view field::escaped_key() const noexcept {
   return std::string_view(reinterpret_cast<const char*>(first.buf), end_quote - first.buf);
 }

+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline std::u8string_view field::escaped_u8key() const noexcept {
+  return internal::as_u8string_view(escaped_key());
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 simdjson_inline value &field::value() & noexcept {
   return second;
 }
@@ -112415,11 +142988,25 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<icelake::ondem
   return first.escaped_key();
 }

+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<icelake::ondemand::field>::escaped_u8key() noexcept {
+  if (error()) { return error(); }
+  return first.escaped_u8key();
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 simdjson_inline simdjson_result<std::string_view> simdjson_result<icelake::ondemand::field>::unescaped_key(bool allow_replacement) noexcept {
   if (error()) { return error(); }
   return first.unescaped_key(allow_replacement);
 }

+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<icelake::ondemand::field>::unescaped_u8key(bool allow_replacement) noexcept {
+  if (error()) { return error(); }
+  return first.unescaped_u8key(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 template<typename string_type>
 simdjson_warn_unused simdjson_inline error_code simdjson_result<icelake::ondemand::field>::unescaped_key(string_type &receiver, bool allow_replacement) noexcept {
   if (error()) { return error(); }
@@ -112463,6 +143050,9 @@ simdjson_inline json_iterator::json_iterator(json_iterator &&other) noexcept
     _depth{other._depth},
     _root{other._root},
     _streaming{other._streaming}
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+    , _allow_incomplete_json{other._allow_incomplete_json}
+#endif
 {
   other.parser = nullptr;
 }
@@ -112474,6 +143064,9 @@ simdjson_inline json_iterator &json_iterator::operator=(json_iterator &&other) n
   _depth = other._depth;
   _root = other._root;
   _streaming = other._streaming;
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  _allow_incomplete_json = other._allow_incomplete_json;
+#endif
   other.parser = nullptr;
   return *this;
 }
@@ -112500,7 +143093,8 @@ simdjson_inline json_iterator::json_iterator(const uint8_t *buf, ondemand::parse
       _string_buf_loc{parser->string_buf.get()},
       _depth{1},
       _root{parser->implementation->structural_indexes.get()},
-      _streaming{streaming}
+      _streaming{streaming},
+      _allow_incomplete_json{true}

 {
   logger::log_headers();
@@ -112572,7 +143166,8 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::skip_child(depth_
 #endif // SIMDJSON_CHECK_EOF
       break;
     case '"':
-      if(*peek() == ':') {
+      // At the end, peek() would read the sentinel, which points into the padding.
+      if(!at_end() && *peek() == ':') {
         // We are at a key!!!
         // This might happen if you just started an object and you skip it immediately.
         // Performance note: it would be nice to get rid of this check as it is somewhat
@@ -112615,7 +143210,7 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::skip_child(depth_
     }
   }

-  return report_error(TAPE_ERROR, "not enough close braces");
+  return report_error(INCOMPLETE_ARRAY_OR_OBJECT, "not enough close braces");
 }

 SIMDJSON_POP_DISABLE_WARNINGS
@@ -112632,6 +143227,17 @@ simdjson_inline bool json_iterator::streaming() const noexcept {
   return _streaming;
 }

+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+simdjson_inline bool json_iterator::allow_incomplete_json() const noexcept {
+  return _allow_incomplete_json;
+}
+
+simdjson_inline size_t json_iterator::remaining_input_length(const uint8_t *json) const noexcept {
+  const uint8_t *end = token.buf + parser->_document_len;
+  return json < end ? size_t(end - json) : 0;
+}
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+
 simdjson_inline token_position json_iterator::root_position() const noexcept {
   return _root;
 }
@@ -112914,7 +143520,7 @@ inline std::ostream& operator<<(std::ostream& out, json_type type) noexcept {
         case json_type::string: out << "string"; break;
         case json_type::boolean: out << "boolean"; break;
         case json_type::null: out << "null"; break;
-        default: SIMDJSON_UNREACHABLE();
+        case json_type::unknown: out << "unknown"; break;
     }
     return out;
 }
@@ -113253,6 +143859,10 @@ inline void log_line(const json_iterator &iter, token_position index, depth_t de
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_iterator.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value-inl.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/jsonpathutil.h" */
+/* amalgamation skipped (editor-only): #include <utility> */
+/* amalgamation skipped (editor-only): #if SIMDJSON_SUPPORTS_CONCEPTS */
+/* amalgamation skipped (editor-only): #include <tuple> // std::forward_as_tuple/get for the variadic for_each adapter */
+/* amalgamation skipped (editor-only): #endif */
 /* amalgamation skipped (editor-only): #if SIMDJSON_STATIC_REFLECTION */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_string_builder.h"  // for constevalutil::fixed_string */
 /* amalgamation skipped (editor-only): #include <meta> */
@@ -113282,12 +143892,21 @@ simdjson_inline simdjson_result<value> object::find_field_unordered(const std::s
   return value(iter.child());
 }
 simdjson_inline simdjson_result<value> object::operator[](const std::string_view key) & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   return find_field_unordered(key);
 }
 simdjson_inline simdjson_result<value> object::operator[](const std::string_view key) && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   return std::forward<object>(*this).find_field_unordered(key);
 }
 simdjson_inline simdjson_result<value> object::find_field(const std::string_view key) & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   bool has_value;
   SIMDJSON_TRY( iter.find_field_raw(key).get(has_value) );
   if (!has_value) {
@@ -113297,6 +143916,9 @@ simdjson_inline simdjson_result<value> object::find_field(const std::string_view
   return value(iter.child());
 }
 simdjson_inline simdjson_result<value> object::find_field(const std::string_view key) && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   bool has_value;
   SIMDJSON_TRY( iter.find_field_raw(key).get(has_value) );
   if (!has_value) {
@@ -113306,6 +143928,150 @@ simdjson_inline simdjson_result<value> object::find_field(const std::string_view
   return value(iter.child());
 }

+#if SIMDJSON_SUPPORTS_CONCEPTS
+template <typename Selector, typename Func>
+  requires key_selector_type<Selector> &&
+           std::is_invocable_v<Func&, std::size_t, value>
+simdjson_flatten simdjson_inline for_each_result object::for_each(Func&& on_match)
+    noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, value>) {
+  // Single pass driven directly by the value_iterator, mirroring
+  // find_field_unordered_raw + value(iter.child()). Compared to walking via
+  // object_iterator/field, this avoids constructing a simdjson_result<field> and
+  // a field (key + value) for every field -- and the development-check bookkeeping
+  // in object_iterator -- building a value only for the fields that actually match.
+  // We operate on a copy of the iterator, as object::begin() would.
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  // Mirror object::begin(): for_each must start at the beginning of the object,
+  // not from some position left behind by a prior find_field on the same object.
+  if (!iter.is_at_iterator_start()) { return {OUT_OF_ORDER_ITERATION, 0}; }
+#endif
+  value_iterator it = iter;
+  std::size_t matched = 0;
+  // Track which selector indices have already matched, as a compile-time bitset
+  // (one 64-bit word per 64 keys): the callback fires once per key, on its first
+  // occurrence, and we stop as soon as every key has matched.
+  constexpr std::size_t seen_words = (Selector::size() + 63) / 64;
+  std::array<std::uint64_t, seen_words> seen{};
+  while (it.is_open()) {
+    raw_json_string key;
+    error_code error;
+    std::size_t idx;
+    if constexpr (Selector::window.ok) {
+      // A window selector confirms a key from its raw bytes alone (the closing
+      // quote bounds it), so we take the length-free path: field_key (no backward
+      // scan, mirroring the ordered find_field) feeding match_raw(raw_json_string).
+      if ((error = it.field_key().get(key))) { it.abandon(); return {error, matched}; }
+      if ((error = it.field_value())) { it.abandon(); return {error, matched}; }
+      idx = Selector::match_raw(key);
+    } else {
+      // Otherwise derive the key length from the structural index (the following
+      // ':' token) to feed the length-aware hash, avoiding a forward SIMD scan.
+      std::size_t key_len;
+      if ((error = it.field_key_with_length(key, key_len))) { it.abandon(); return {error, matched}; }
+      if ((error = it.field_value())) { it.abandon(); return {error, matched}; }
+      idx = Selector::match_raw(key.raw(), key_len);
+    }
+    if (idx < Selector::size()) {
+      const std::uint64_t seen_bit = std::uint64_t{1} << (idx & 63);
+      std::uint64_t &seen_word = seen[idx >> 6];
+      if (!(seen_word & seen_bit)) {
+        seen_word |= seen_bit;
+        value matched_value(it.child());
+        // The callback may return void or anything convertible to error_code
+        // (error_code itself, or a for_each_result from a nested for_each). When
+        // it yields an error_code, we stop at the first non-SUCCESS result and
+        // propagate it so the caller can surface value-parse errors (for example,
+        // a type mismatch on a matched field). A void-returning callback is
+        // responsible for handling its own errors.
+        if constexpr (std::is_convertible_v<decltype(on_match(idx, matched_value)), error_code>) {
+          // Unlike the internal-error paths above, a callback error does not
+          // abandon the iterator: we leave it recoverable so the caller can keep
+          // using the object (or its parent) after handling the error.
+          if ((error = on_match(idx, matched_value))) { return {error, matched}; }
+        } else {
+          on_match(idx, matched_value);
+        }
+        if (++matched >= Selector::size()) { break; }
+      }
+    }
+    // Mirror object_iterator::operator++'s safety rail: if the callback consumed
+    // the value and left the iterator closed or in error (e.g. a void callback
+    // that swallowed a fatal sub-iteration error), stop here rather than calling
+    // skip_child on a closed iterator.
+    if (!it.is_open()) { break; }
+    // Skip the value (a no-op if the callback consumed it) and step to the next
+    // field; has_next_field() ends the container on '}', which closes the loop.
+    if ((error = it.skip_child())) { it.abandon(); return {error, matched}; }
+    if ((error = it.has_next_field().error())) { return {error, matched}; }
+  }
+  return {SUCCESS, matched};
+}
+
+namespace key_selector_for_each_detail {
+
+// Dispatch a matched value to the I-th handler in the tuple (0-based). A handler
+// is either an invocable taking the value (void- or error_code-returning) or a
+// deserialization target, in which case we assign via value::get. We always
+// return error_code so the core (index, value) for_each can treat the adapter
+// uniformly.
+template <typename Tuple, std::size_t... Is>
+simdjson_really_inline error_code dispatch_value(
+    std::size_t idx, Tuple& handlers, value v, std::index_sequence<Is...>) {
+  error_code err = SUCCESS;
+  auto try_one = [&](auto Ic) {
+    constexpr std::size_t I = decltype(Ic)::value;
+    if (idx == I) {
+      auto&& h = std::get<I>(handlers);
+      using H = std::remove_reference_t<decltype(h)>;
+      if constexpr (std::is_invocable_v<H&, value>) {
+        // A handler returning void runs for its side effects; one returning
+        // anything convertible to error_code (error_code, or a for_each_result
+        // from a nested for_each) has its error captured and propagated.
+        if constexpr (std::is_convertible_v<decltype(h(v)), error_code>) {
+          err = h(v);
+        } else {
+          h(v);
+        }
+      } else {
+        // Direct deserialization target: assign the matched value into it.
+        err = v.get(h);
+      }
+    }
+  };
+  (try_one(std::integral_constant<std::size_t, Is>{}), ...);
+  return err;
+}
+
+} // namespace key_selector_for_each_detail
+
+template <typename Selector, typename... Handlers>
+  requires key_selector_type<Selector> &&
+           (sizeof...(Handlers) == Selector::size()) &&
+           (key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_flatten simdjson_inline for_each_result object::for_each(Handlers&&... on_match)
+    noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+  auto handlers = std::forward_as_tuple(std::forward<Handlers>(on_match)...);
+  // Reuse the single (index, value) implementation via a tiny adapter.
+  // The adapter is called once per *matched* key (very few); the hot path
+  // (iteration + match_raw + seen bitset) stays exactly the same.
+  return this->template for_each<Selector>(
+      [&](std::size_t i, value v) -> error_code {
+        return key_selector_for_each_detail::dispatch_value(
+            i, handlers, v, std::make_index_sequence<Selector::size()>{});
+      });
+}
+
+template <constevalutil::fixed_string... Keys, typename... Handlers>
+  requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+           (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+           (key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline for_each_result object::for_each(Handlers&&... on_match)
+    noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+  using Selector = key_selector<Keys...>;
+  return this->template for_each<Selector>(std::forward<Handlers>(on_match)...);
+}
+#endif
+
 simdjson_inline simdjson_result<object> object::start(value_iterator &iter) noexcept {
   SIMDJSON_TRY( iter.start_object().error() );
   return object(iter);
@@ -113341,6 +144107,9 @@ simdjson_warn_unused simdjson_inline error_code object::consume() noexcept {
 }

 simdjson_inline simdjson_result<std::string_view> object::raw_json() noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   const uint8_t * starting_point{iter.peek_start()};
   auto error = consume();
   if(error) { return error; }
@@ -113362,9 +144131,17 @@ simdjson_inline object::object(const value_iterator &_iter) noexcept
 {
 }

-simdjson_inline simdjson_result<object_iterator> object::begin() noexcept {
+simdjson_inline simdjson_result<object_iterator> object::begin() & noexcept {
 #if SIMDJSON_DEVELOPMENT_CHECKS
-  if (!iter.is_at_iterator_start()) { return OUT_OF_ORDER_ITERATION; }
+  if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
+  return object_iterator(iter, this);
+#endif
+  return object_iterator(iter);
+}
+simdjson_inline simdjson_result<object_iterator> object::begin() && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  // The object is a temporary that the iterator may outlive: do not lock it.
+  if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
 #endif
   return object_iterator(iter);
 }
@@ -113373,7 +144150,9 @@ simdjson_inline simdjson_result<object_iterator> object::end() noexcept {
 }

 inline simdjson_result<value> object::at_pointer(std::string_view json_pointer) noexcept {
-  if (json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
+  // An empty pointer has no json_pointer[0]: with a default-constructed
+  // std::string_view, reading it dereferences a null pointer.
+  if (json_pointer.empty() || json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
   json_pointer = json_pointer.substr(1);
   size_t slash = json_pointer.find('/');
   std::string_view key = json_pointer.substr(0, slash);
@@ -113475,6 +144254,9 @@ simdjson_inline simdjson_result<size_t> object::count_fields() & noexcept {
 }

 simdjson_inline simdjson_result<bool> object::is_empty() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   bool is_not_empty;
   auto error = iter.reset_object().get(is_not_empty);
   if(error) { return error; }
@@ -113482,9 +144264,41 @@ simdjson_inline simdjson_result<bool> object::is_empty() & noexcept {
 }

 simdjson_inline simdjson_result<bool> object::reset() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   return iter.reset_object();
 }

+simdjson_inline object_position object::get_current_position() const noexcept {
+  return object_position{iter.position(), iter.json_iter().depth()};
+}
+
+simdjson_inline error_code object::revert_position(object_position position) noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
+  // json_iterator::reenter_child() requires the live depth to be exactly
+  // one level shallower than the target (matching how every other depth
+  // transition in this iterator works), and under SIMDJSON_DEVELOPMENT_CHECKS
+  // additionally validates against the parser's per-depth container-start
+  // bookkeeping. Neither applies here: depending on what was captured and
+  // what has happened since (a scalar field fully consumed, a compound
+  // value left open, a find_field() miss that scanned past everything),
+  // the live depth when reverting can be any number of levels away from
+  // the captured one, and the captured depth is not necessarily a
+  // container's own start. reenter_at() moves directly, matching how
+  // reset_object() itself repositions without going through reenter_child().
+  iter.reenter_at(position.position, position.depth);
+  return SUCCESS;
+}
+
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline void object::set_locked(bool _locked) noexcept {
+  locked = _locked;
+}
+#endif
+
 #if SIMDJSON_SUPPORTS_CONCEPTS && SIMDJSON_STATIC_REFLECTION

 template<constevalutil::fixed_string... FieldNames, typename T>
@@ -113542,10 +144356,14 @@ simdjson_inline simdjson_result<icelake::ondemand::object>::simdjson_result(icel
 simdjson_inline simdjson_result<icelake::ondemand::object>::simdjson_result(error_code error) noexcept
     : implementation_simdjson_result_base<icelake::ondemand::object>(error) {}

-simdjson_inline simdjson_result<icelake::ondemand::object_iterator> simdjson_result<icelake::ondemand::object>::begin() noexcept {
+simdjson_inline simdjson_result<icelake::ondemand::object_iterator> simdjson_result<icelake::ondemand::object>::begin() & noexcept {
   if (error()) { return error(); }
   return first.begin();
 }
+simdjson_inline simdjson_result<icelake::ondemand::object_iterator> simdjson_result<icelake::ondemand::object>::begin() && noexcept {
+  if (error()) { return error(); }
+  return std::move(first).begin();
+}
 simdjson_inline simdjson_result<icelake::ondemand::object_iterator> simdjson_result<icelake::ondemand::object>::end() noexcept {
   if (error()) { return error(); }
   return first.end();
@@ -113599,11 +144417,55 @@ simdjson_inline error_code simdjson_result<icelake::ondemand::object>::for_each_
   return first.for_each_at_path_with_wildcard(json_path, std::forward<Func>(callback));
 }

+#if SIMDJSON_SUPPORTS_CONCEPTS
+template <typename Selector, typename Func>
+  requires icelake::ondemand::key_selector_type<Selector> &&
+           std::is_invocable_v<Func&, std::size_t, icelake::ondemand::value>
+simdjson_inline icelake::ondemand::for_each_result
+simdjson_result<icelake::ondemand::object>::for_each(Func&& on_match)
+    noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, icelake::ondemand::value>) {
+  if (error()) { return {error(), 0}; }
+  return first.template for_each<Selector>(std::forward<Func>(on_match));
+}
+
+template <typename Selector, typename... Handlers>
+  requires icelake::ondemand::key_selector_type<Selector> &&
+           (sizeof...(Handlers) == Selector::size()) &&
+           (icelake::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline icelake::ondemand::for_each_result
+simdjson_result<icelake::ondemand::object>::for_each(Handlers&&... on_match)
+    noexcept(icelake::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+  if (error()) { return {error(), 0}; }
+  return first.template for_each<Selector>(std::forward<Handlers>(on_match)...);
+}
+
+template <constevalutil::fixed_string... Keys, typename... Handlers>
+  requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+           (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+           (icelake::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline icelake::ondemand::for_each_result
+simdjson_result<icelake::ondemand::object>::for_each(Handlers&&... on_match)
+    noexcept(icelake::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+  if (error()) { return {error(), 0}; }
+  return first.template for_each<Keys...>(std::forward<Handlers>(on_match)...);
+}
+#endif
+
 inline simdjson_result<bool> simdjson_result<icelake::ondemand::object>::reset() noexcept {
   if (error()) { return error(); }
   return first.reset();
 }

+inline simdjson_result<icelake::ondemand::object_position> simdjson_result<icelake::ondemand::object>::get_current_position() noexcept {
+  if (error()) { return error(); }
+  return first.get_current_position();
+}
+
+inline error_code simdjson_result<icelake::ondemand::object>::revert_position(icelake::ondemand::object_position position) noexcept {
+  if (error()) { return error(); }
+  return first.revert_position(position);
+}
+
 inline simdjson_result<bool> simdjson_result<icelake::ondemand::object>::is_empty() noexcept {
   if (error()) { return error(); }
   return first.is_empty();
@@ -113647,6 +144509,61 @@ simdjson_inline object_iterator::object_iterator(const value_iterator &_iter) no
   : iter{_iter}
 {}

+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline object_iterator::object_iterator(const value_iterator &_iter, object* _parent) noexcept
+  : parent{_parent}, iter{_iter}
+{
+  if (parent) parent->set_locked(true);
+}
+#endif
+
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline object_iterator::~object_iterator() noexcept
+{
+  if (parent) parent->set_locked(false);
+}
+
+simdjson_inline object_iterator::object_iterator(object_iterator&& other) noexcept
+  : has_been_referenced{other.has_been_referenced},
+    parent{other.parent},
+    iter{std::move(other.iter)}
+{
+  other.parent = nullptr;
+}
+
+simdjson_inline object_iterator& object_iterator::operator=(object_iterator&& other) noexcept {
+  if (this != &other)
+  {
+    if (parent)
+      parent->set_locked(false);
+    has_been_referenced = other.has_been_referenced;
+    parent = other.parent;
+    iter = std::move(other.iter);
+
+    other.parent = nullptr;
+  }
+  return *this;
+}
+
+simdjson_inline object_iterator::object_iterator(const object_iterator& other) noexcept
+  : has_been_referenced{other.has_been_referenced},
+    parent{nullptr},
+    iter{other.iter}
+{}
+
+simdjson_inline object_iterator& object_iterator::operator=(const object_iterator& other) noexcept {
+  if (this != &other)
+  {
+    if (parent)
+      parent->set_locked(false);
+    has_been_referenced = other.has_been_referenced;
+    parent = nullptr;
+    iter = other.iter;
+  }
+  return *this;
+}
+#endif
+
 simdjson_inline simdjson_result<field> object_iterator::operator*() noexcept {
 #if SIMDJSON_DEVELOPMENT_CHECKS
   // We must call * once per iteration.
@@ -113774,6 +144691,147 @@ simdjson_inline simdjson_result<icelake::ondemand::object_iterator> &simdjson_re

 #endif // SIMDJSON_GENERIC_ONDEMAND_OBJECT_ITERATOR_INL_H
 /* end file simdjson/generic/ondemand/object_iterator-inl.h for icelake */
+/* including simdjson/generic/ondemand/ranges-inl.h for icelake: #include "simdjson/generic/ondemand/ranges-inl.h" */
+/* begin file simdjson/generic/ondemand/ranges-inl.h for icelake */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/ranges.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array_iterator-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object_iterator-inl.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#if SIMDJSON_SUPPORTS_RANGES
+
+namespace simdjson {
+namespace icelake {
+namespace ondemand {
+
+//
+// array_range_iterator
+//
+
+simdjson_inline array_range_iterator::array_range_iterator(array_iterator iter) noexcept
+  : iter_{iter} {}
+
+simdjson_inline simdjson_result<value> array_range_iterator::operator*() const noexcept {
+  return *iter_;
+}
+
+simdjson_inline array_range_iterator& array_range_iterator::operator++() noexcept {
+  ++iter_;
+  return *this;
+}
+
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+simdjson_inline void array_range_iterator::operator++(int) noexcept {
+  ++*this;
+}
+SIMDJSON_POP_DISABLE_WARNINGS
+
+//
+// array_range
+//
+
+simdjson_inline array_range::array_range(array& arr) noexcept {
+  auto b = arr.begin();
+  if (b.error()) { error_ = b.error(); return; }
+  begin_ = b.value_unsafe();
+  end_ = arr.end().value_unsafe();
+}
+
+simdjson_inline array_range_iterator array_range::begin() noexcept {
+  return array_range_iterator(begin_);
+}
+
+simdjson_inline array_range_iterator array_range::end() noexcept {
+  return array_range_iterator(end_);
+}
+
+//
+// object_range_iterator
+//
+
+simdjson_inline object_range_iterator::object_range_iterator(object_iterator iter) noexcept
+  : iter_{iter} {}
+
+simdjson_inline simdjson_result<field> object_range_iterator::operator*() const noexcept {
+  return *iter_;
+}
+
+simdjson_inline object_range_iterator& object_range_iterator::operator++() noexcept {
+  ++iter_;
+  return *this;
+}
+
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+simdjson_inline void object_range_iterator::operator++(int) noexcept {
+  ++*this;
+}
+SIMDJSON_POP_DISABLE_WARNINGS
+
+
+//
+// object_range
+//
+
+simdjson_inline object_range::object_range(object& obj) noexcept {
+  auto b = obj.begin();
+  if (b.error()) { error_ = b.error(); return; }
+  begin_ = b.value_unsafe();
+  end_ = obj.end().value_unsafe();
+}
+
+simdjson_inline object_range_iterator object_range::begin() noexcept {
+  return object_range_iterator(begin_);
+}
+
+simdjson_inline object_range_iterator object_range::end() noexcept {
+  return object_range_iterator(end_);
+}
+
+//
+// Free functions
+//
+
+simdjson_inline array_range get_range(array& arr) noexcept {
+  return array_range(arr);
+}
+
+simdjson_inline object_range get_key_value_range(object& obj) noexcept {
+  return object_range(obj);
+}
+
+#if SIMDJSON_EXCEPTIONS
+simdjson_inline array_range get_range(simdjson_result<array> result) {
+  return array_range(result.value());
+}
+
+simdjson_inline object_range get_key_value_range(simdjson_result<object> result) {
+  return object_range(result.value());
+}
+#endif // SIMDJSON_EXCEPTIONS
+
+} // namespace ondemand
+} // namespace icelake
+} // namespace simdjson
+
+// Verify the range wrapper types satisfy the expected C++20 concepts.
+static_assert(std::input_iterator<simdjson::icelake::ondemand::array_range_iterator>);
+static_assert(std::input_iterator<simdjson::icelake::ondemand::object_range_iterator>);
+static_assert(std::ranges::input_range<simdjson::icelake::ondemand::array_range>);
+static_assert(std::ranges::input_range<simdjson::icelake::ondemand::object_range>);
+static_assert(std::ranges::view<simdjson::icelake::ondemand::array_range>);
+static_assert(std::ranges::view<simdjson::icelake::ondemand::object_range>);
+
+#endif // SIMDJSON_SUPPORTS_RANGES
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H
+/* end file simdjson/generic/ondemand/ranges-inl.h for icelake */
 /* including simdjson/generic/ondemand/parser-inl.h for icelake: #include "simdjson/generic/ondemand/parser-inl.h" */
 /* begin file simdjson/generic/ondemand/parser-inl.h for icelake */
 #ifndef SIMDJSON_GENERIC_ONDEMAND_PARSER_INL_H
@@ -113805,7 +144863,10 @@ simdjson_warn_unused simdjson_inline error_code parser::allocate(size_t new_capa

   // string_capacity copied from document::allocate
   _capacity = 0;
-  size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * new_capacity / 3 + SIMDJSON_PADDING, 64);
+  if(5 * (new_capacity / 3) + SIMDJSON_PADDING < SIMDJSON_PADDING) {
+    return CAPACITY; // overflow, only happen on legacy 32-bit systems with very large capacity
+  }
+  size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * (new_capacity / 3) + SIMDJSON_PADDING, 64);
   string_buf.reset(new (std::nothrow) uint8_t[string_capacity]);
 #if SIMDJSON_DEVELOPMENT_CHECKS
   start_positions.reset(new (std::nothrow) token_position[new_max_depth]);
@@ -113830,6 +144891,7 @@ simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate(p
   if (!json.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }

   json.remove_utf8_bom();
+  _document_len = json.length();

   // Allocate if needed
   if (capacity() < json.length() || !string_buf) {
@@ -113846,6 +144908,7 @@ simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate_a
   if (!json.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }

   json.remove_utf8_bom();
+  _document_len = json.length();

   // Allocate if needed
   if (capacity() < json.length() || !string_buf) {
@@ -113911,6 +144974,34 @@ simdjson_warn_unused simdjson_inline simdjson_result<json_iterator> parser::iter
   return json_iterator(reinterpret_cast<const uint8_t *>(json.data()), this);
 }

+inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size) noexcept {
+  if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
+  if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+    buf += 3;
+    len -= 3;
+  }
+  return document_stream(*this, buf, len, batch_size, false, stream_format::whitespace_delimited);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size) noexcept {
+  return iterate_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(padded_string_view s, size_t batch_size) noexcept {
+  if (!s.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
+  return iterate_many(s.data(), s.length(), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const padded_string &s, size_t batch_size) noexcept {
+  return iterate_many(padded_string_view(s), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const std::string &s, size_t batch_size) noexcept {
+  return iterate_many(padded_string_view(s), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size) noexcept {
+  return iterate_many(pad(s), batch_size);
+}
+
+#ifndef SIMDJSON_DISABLE_DEPRECATED_API
+SIMDJSON_PUSH_DISABLE_WARNINGS
+SIMDJSON_DISABLE_DEPRECATED_WARNING
 inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept {
   // Warning: no check is done on the buffer padding. We trust the user.
   if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
@@ -113918,8 +145009,11 @@ inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf,
     buf += 3;
     len -= 3;
   }
-  if(allow_comma_separated && batch_size < len) { batch_size = len; }
-  return document_stream(*this, buf, len, batch_size, allow_comma_separated);
+  // Map allow_comma_separated to stream_format::comma_delimited
+  if (allow_comma_separated) {
+    return document_stream(*this, buf, len, batch_size, false, stream_format::comma_delimited);
+  }
+  return document_stream(*this, buf, len, batch_size, false, stream_format::whitespace_delimited);
 }

 inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept {
@@ -113939,6 +145033,51 @@ inline simdjson_result<document_stream> parser::iterate_many(const std::string &
 inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size, bool allow_comma_separated) noexcept {
   return iterate_many(pad(s), batch_size, allow_comma_separated);
 }
+SIMDJSON_POP_DISABLE_WARNINGS
+#endif // SIMDJSON_DISABLE_DEPRECATED_API
+
+inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+  if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
+  if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+    buf += 3;
+    len -= 3;
+  }
+  if (format == stream_format::comma_delimited_array) {
+    // Strip leading JSON whitespace.
+    while (len > 0 && (buf[0] == ' ' || buf[0] == '\t' || buf[0] == '\n' || buf[0] == '\r')) {
+      buf++; len--;
+    }
+    // Expect the opening '['.
+    if (len == 0 || buf[0] != '[') { return TAPE_ERROR; }
+    buf++; len--;
+    // Strip trailing JSON whitespace.
+    while (len > 0 && (buf[len-1] == ' ' || buf[len-1] == '\t' || buf[len-1] == '\n' || buf[len-1] == '\r')) {
+      len--;
+    }
+    // Expect the closing ']'.
+    if (len == 0 || buf[len-1] != ']') { return TAPE_ERROR; }
+    len--;
+    // Fall through to comma_delimited over the array contents.
+    format = stream_format::comma_delimited;
+  }
+  return document_stream(*this, buf, len, batch_size, false, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+  return iterate_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(padded_string_view s, size_t batch_size, stream_format format) noexcept {
+  if (!s.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
+  return iterate_many(s.data(), s.length(), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const std::string &s, size_t batch_size, stream_format format) noexcept {
+  return iterate_many(padded_string_view(s), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size, stream_format format) noexcept {
+  return iterate_many(pad(s), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const padded_string &s, size_t batch_size, stream_format format) noexcept {
+  return iterate_many(padded_string_view(s), batch_size, format);
+}
 simdjson_pure simdjson_inline size_t parser::capacity() const noexcept {
   return _capacity;
 }
@@ -114346,6 +145485,27 @@ namespace simdjson {
 namespace icelake {
 namespace ondemand {

+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+simdjson_inline bool raw_json_string_is_quote_terminated(const uint8_t *json, uint32_t max_len) noexcept {
+  bool escaping{false};
+  for (uint32_t i = 1; i < max_len; i++) {
+    switch (json[i]) {
+      case '"':
+        if (!escaping) { return true; }
+        escaping = false;
+        break;
+      case '\\':
+        escaping = !escaping;
+        break;
+      default:
+        escaping = false;
+        break;
+    }
+  }
+  return false;
+}
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+
 simdjson_inline value_iterator::value_iterator(
   json_iterator *json_iter,
   depth_t depth,
@@ -114733,6 +145893,22 @@ simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iter
   return raw_json_string(key);
 }

+simdjson_warn_unused simdjson_inline error_code value_iterator::field_key_with_length(raw_json_string &key, std::size_t &len) noexcept {
+  assert_at_next();
+
+  const uint8_t *k = _json_iter->return_current_and_advance();
+  if (*(k++) != '"') { return report_error(TAPE_ERROR, "Object key is not a string"); }
+  // After return_current_and_advance(), the current token is the ':' that follows
+  // the key. The closing quote sits just before it (only JSON whitespace may
+  // intervene), so step back from the ':' to the closing quote to get the length.
+  // In minified JSON this is a single back-step.
+  const char *q = reinterpret_cast<const char *>(_json_iter->peek());
+  do { --q; } while (*q != '"');
+  key = raw_json_string(k);
+  len = static_cast<std::size_t>(q - reinterpret_cast<const char *>(k));
+  return SUCCESS;
+}
+
 simdjson_warn_unused simdjson_inline error_code value_iterator::field_value() noexcept {
   assert_at_next();

@@ -114850,7 +146026,7 @@ simdjson_warn_unused simdjson_inline error_code value_iterator::get_string(strin
   auto saved_string_buf_loc = _json_iter->string_buf_loc();
   auto err = get_string(allow_replacement).get(content);
   if (err) { return err; }
-  receiver = content;
+  internal::assign_utf8(receiver, content);
   // Restore the string buffer location, effectively discarding any temporary string storage
   _json_iter->string_buf_loc() = saved_string_buf_loc;
   return SUCCESS;
@@ -114861,6 +146037,15 @@ simdjson_warn_unused simdjson_inline simdjson_result<std::string_view> value_ite
 simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iterator::get_raw_json_string() noexcept {
   auto json = peek_scalar("string");
   if (*json != '"') { return incorrect_type_error("Not a string"); }
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  if (_json_iter->allow_incomplete_json()) {
+    const size_t remaining_input_length = _json_iter->remaining_input_length(json);
+    const uint32_t max_len = remaining_input_length < peek_start_length() ? uint32_t(remaining_input_length) : peek_start_length();
+    if (!raw_json_string_is_quote_terminated(json, max_len)) {
+      return STRING_ERROR;
+    }
+  }
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
   advance_scalar("string");
   return raw_json_string(json+1);
 }
@@ -114894,6 +146079,16 @@ simdjson_warn_unused simdjson_inline simdjson_result<double> value_iterator::get
   if(result.error() == SUCCESS) { advance_non_root_scalar("double"); }
   return result;
 }
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_float() noexcept {
+  auto result = numberparsing::parse_float(peek_non_root_scalar("float"));
+  if(result.error() == SUCCESS) { advance_non_root_scalar("float"); }
+  return result;
+}
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_float_in_string() noexcept {
+  auto result = numberparsing::parse_float_in_string(peek_non_root_scalar("float"));
+  if(result.error() == SUCCESS) { advance_non_root_scalar("float"); }
+  return result;
+}
 simdjson_warn_unused simdjson_inline simdjson_result<bool> value_iterator::get_bool() noexcept {
   auto result = parse_bool(peek_non_root_scalar("bool"));
   if(result.error() == SUCCESS) { advance_non_root_scalar("bool"); }
@@ -114996,7 +146191,7 @@ simdjson_warn_unused simdjson_inline error_code value_iterator::get_root_string(
   auto saved_string_buf_loc = _json_iter->string_buf_loc();
   auto err = get_root_string(check_trailing, allow_replacement).get(content);
   if (err) { return err; }
-  receiver = content;
+  internal::assign_utf8(receiver, content);
   // Restore the string buffer location, effectively discarding any temporary string storage
   _json_iter->string_buf_loc() = saved_string_buf_loc;
   return SUCCESS;
@@ -115008,6 +146203,15 @@ simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iter
   auto json = peek_scalar("string");
   if (*json != '"') { return incorrect_type_error("Not a string"); }
   if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  if (_json_iter->allow_incomplete_json()) {
+    const size_t remaining_input_length = _json_iter->remaining_input_length(json);
+    const uint32_t max_len = remaining_input_length < peek_root_length() ? uint32_t(remaining_input_length) : peek_root_length();
+    if (!raw_json_string_is_quote_terminated(json, max_len)) {
+      return STRING_ERROR;
+    }
+  }
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
   advance_scalar("string");
   return raw_json_string(json+1);
 }
@@ -115117,6 +146321,43 @@ simdjson_warn_unused simdjson_inline simdjson_result<double> value_iterator::get
   return result;
 }

+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_root_float(bool check_trailing) noexcept {
+  auto max_len = peek_root_length();
+  auto json = peek_root_scalar("float");
+  // We use the same buffer size as get_root_double: the number of significant
+  // digits that matter is smaller for binary32, but the JSON document may still
+  // spell out a long number that we must parse (and round) faithfully.
+  uint8_t tmpbuf[1074+8+1+1]; // +1 for null termination.
+  tmpbuf[1074+8+1] = '\0'; // make sure that buffer is always null terminated.
+  if (!_json_iter->copy_to_buffer(json, max_len, tmpbuf, 1074+8+1)) {
+    logger::log_error(*_json_iter, start_position(), depth(), "Root number more than 1082 characters");
+    return NUMBER_ERROR;
+  }
+  auto result = numberparsing::parse_float(tmpbuf);
+  if(result.error() == SUCCESS) {
+    if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+    advance_root_scalar("float");
+  }
+  return result;
+}
+
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_root_float_in_string(bool check_trailing) noexcept {
+  auto max_len = peek_root_length();
+  auto json = peek_root_scalar("float");
+  uint8_t tmpbuf[1074+8+1+1]; // +1 for null termination.
+  tmpbuf[1074+8+1] = '\0'; // make sure that buffer is always null terminated.
+  if (!_json_iter->copy_to_buffer(json, max_len, tmpbuf, 1074+8+1)) {
+    logger::log_error(*_json_iter, start_position(), depth(), "Root number more than 1082 characters");
+    return NUMBER_ERROR;
+  }
+  auto result = numberparsing::parse_float_in_string(tmpbuf);
+  if(result.error() == SUCCESS) {
+    if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+    advance_root_scalar("float");
+  }
+  return result;
+}
+
 simdjson_warn_unused simdjson_inline simdjson_result<bool> value_iterator::get_root_bool(bool check_trailing) noexcept {
   auto max_len = peek_root_length();
   auto json = peek_root_scalar("bool");
@@ -115345,6 +146586,21 @@ simdjson_inline void value_iterator::move_at_container_start() noexcept {
   _json_iter->token.set_position(_start_position + 1);
 }

+simdjson_inline void value_iterator::reenter_at(token_position position, depth_t depth) noexcept {
+  // Unlike reenter_child(), this does not require the live depth to be
+  // exactly one level shallower than depth, nor does it validate against
+  // the parser's per-depth container-start bookkeeping: neither holds in
+  // general for a caller-supplied snapshot (see object_position). What
+  // must still always hold, regardless of what was captured or how far
+  // the live iterator has since moved, is that position and depth are
+  // themselves sane values -- this is the same bound reenter_child()
+  // itself applies unconditionally.
+  SIMDJSON_ASSUME(position != nullptr);
+  SIMDJSON_ASSUME(depth >= 1 && depth < INT32_MAX);
+  _json_iter->_depth = depth;
+  _json_iter->token.set_position(position);
+}
+
 simdjson_inline simdjson_result<bool> value_iterator::reset_array() noexcept {
   if(error()) { return error(); }
   move_at_container_start();
@@ -116766,16 +148022,6 @@ simdjson_inline int count_ones(uint64_t input_num) {
 }
 #endif

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
-                                         uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
-  *result = value1 + value2;
-  return *result < value1;
-#else
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-#endif
-}

 } // unnamed namespace
 } // namespace ppc64
@@ -117481,7 +148727,7 @@ simdjson_inline escaping escaping::copy_and_find(const uint8_t *src, uint8_t *ds
 /* end file simdjson/ppc64/begin.h */
 /* including simdjson/generic/ondemand/amalgamated.h for ppc64: #include "simdjson/generic/ondemand/amalgamated.h" */
 /* begin file simdjson/generic/ondemand/amalgamated.h for ppc64 */
-#if defined(SIMDJSON_CONDITIONAL_INCLUDE) && !defined(SIMDJSON_GENERIC_BUILDER_DEPENDENCIES_H)
+#if defined(SIMDJSON_CONDITIONAL_INCLUDE) && !defined(SIMDJSON_GENERIC_ONDEMAND_DEPENDENCIES_H)
 #error simdjson/generic/ondemand/dependencies.h must be included before simdjson/generic/ondemand/amalgamated.h!
 #endif

@@ -117530,6 +148776,13 @@ class token_iterator;
 class value;
 class value_iterator;

+#if SIMDJSON_SUPPORTS_RANGES
+class array_range;
+class array_range_iterator;
+class object_range;
+class object_range_iterator;
+#endif // SIMDJSON_SUPPORTS_RANGES
+
 } // namespace ondemand
 } // namespace ppc64
 } // namespace simdjson
@@ -117562,6 +148815,9 @@ template <> struct is_builtin_deserializable<ppc64::ondemand::object> : std::tru
 template <> struct is_builtin_deserializable<ppc64::ondemand::value> : std::true_type {};
 template <> struct is_builtin_deserializable<ppc64::ondemand::raw_json_string> : std::true_type {};
 template <> struct is_builtin_deserializable<std::string_view> : std::true_type {};
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template <> struct is_builtin_deserializable<std::u8string_view> : std::true_type {};
+#endif // SIMDJSON_SUPPORTS_CHAR8_T

 template <typename T>
 concept is_builtin_deserializable_v = is_builtin_deserializable<T>::value;
@@ -117579,6 +148835,10 @@ concept nothrow_custom_deserializable = nothrow_tag_invocable<deserialize_tag, V
 template <typename T, typename ValT = ppc64::ondemand::value>
 concept nothrow_deserializable = nothrow_custom_deserializable<T, ValT> || is_builtin_deserializable_v<T>;

+// True iff get<T>() on ValT cannot throw: no custom tag_invoke, or that tag_invoke is noexcept.
+template <typename T, typename ValT = ppc64::ondemand::value>
+concept nothrow_gettable = !custom_deserializable<T, ValT> || nothrow_custom_deserializable<T, ValT>;
+
 /// Deserialize Tag
 inline constexpr struct deserialize_tag {
   using array_type = ppc64::ondemand::array;
@@ -117793,6 +149053,17 @@ public:
    */
   simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> field_key() noexcept;

+  /**
+   * Get the current field's key together with its raw byte length.
+   *
+   * Like field_key(), but also returns the number of raw key bytes (the distance
+   * from the first key byte to the closing quote). The length is recovered from
+   * the structural index -- the next structural token is the ':' -- by stepping
+   * back over any whitespace to the closing quote, avoiding a forward SIMD scan
+   * for the closing quote. Leaves the iterator positioned exactly as field_key().
+   */
+  simdjson_warn_unused simdjson_inline error_code field_key_with_length(raw_json_string &key, std::size_t &len) noexcept;
+
   /**
    * Pass the : in the field and move to its value.
    */
@@ -117945,6 +149216,8 @@ public:
   simdjson_warn_unused simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_warn_unused simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_warn_unused simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<bool> get_bool() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<bool> is_null() noexcept;
   simdjson_warn_unused simdjson_inline bool is_negative() noexcept;
@@ -117963,6 +149236,8 @@ public:
   simdjson_warn_unused simdjson_inline simdjson_result<int64_t> get_root_int64_in_string(bool check_trailing) noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<double> get_root_double(bool check_trailing) noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<double> get_root_double_in_string(bool check_trailing) noexcept;
+  simdjson_warn_unused simdjson_inline simdjson_result<float> get_root_float(bool check_trailing) noexcept;
+  simdjson_warn_unused simdjson_inline simdjson_result<float> get_root_float_in_string(bool check_trailing) noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<bool> get_root_bool(bool check_trailing) noexcept;
   simdjson_warn_unused simdjson_inline bool is_root_negative() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<bool> is_root_integer(bool check_trailing) noexcept;
@@ -118098,6 +149373,15 @@ protected:

   /** @copydoc error_code json_iterator::position() const noexcept; */
   simdjson_inline token_position position() const noexcept;
+  /**
+   * Move the live iterator directly to the given position and depth, without
+   * validating against the parser's per-depth container-start bookkeeping
+   * (unlike json_iterator::reenter_child()). Used to restore a previously
+   * captured mid-container position (see object::revert_position()): that
+   * bookkeeping only tracks each container's own start, not every position
+   * a caller might later capture and revert to, so it does not apply here.
+   */
+  simdjson_inline void reenter_at(token_position position, depth_t depth) noexcept;
   /** @copydoc error_code json_iterator::end_position() const noexcept; */
   simdjson_inline token_position last_position() const noexcept;
   /** @copydoc error_code json_iterator::end_position() const noexcept; */
@@ -118166,9 +149450,12 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+   * Supported types: object, array, raw_json_string, string_view, u8string_view (C++20), uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool
    *
-   * You may use get_double(), get_bool(), get_uint64(), get_int64(),
+   * You may use get_double(), get_float(), get_bool(), get_uint64(), get_int64(),
+   * get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(),
+   * get_float32(), get_float64(),
    * get_object(), get_array(), get_raw_json_string(), or get_string() instead.
    * When SIMDJSON_SUPPORTS_CONCEPTS is set, custom types are also supported.
    *
@@ -118178,7 +149465,7 @@ public:
   template <typename T>
   simdjson_inline simdjson_result<T> get()
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+    noexcept(nothrow_gettable<T, value>)
 #else
     noexcept
 #endif
@@ -118193,7 +149480,8 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+   * Supported types: object, array, raw_json_string, string_view, u8string_view (C++20), uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool
    * If the macro SIMDJSON_SUPPORTS_CONCEPTS is set, then custom types are also supported.
    *
    * @param out This is set to a value of the given type, parsed from the JSON. If there is an error, this may not be initialized.
@@ -118203,7 +149491,7 @@ public:
   template <typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out)
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+    noexcept(nothrow_gettable<T, value>)
 #else
     noexcept
 #endif
@@ -118231,7 +149519,7 @@ public:
       "And you do not seem to have added support for it. Indeed, we have that "
       "simdjson::custom_deserializable<T> is false and the type T is not a default type "
       "such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-      "int64_t, double, or bool.");
+      "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
     static_cast<void>(out); // to get rid of unused errors
     return UNINITIALIZED;
   }
@@ -118240,7 +149528,8 @@ public:
     // immediately fail.
     static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
       "The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-      "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+      "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+      "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
       " get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
       " You may also add support for custom types, see our documentation.");
     static_cast<void>(out); // to get rid of unused errors
@@ -118318,6 +149607,50 @@ public:
    */
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;

+  /**
+   * Cast this JSON value to a 16-bit unsigned integer.
+   *
+   * Calls get_uint64() and checks that the result fits in a uint16_t.
+   *
+   * @returns A 16-bit unsigned integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint16_t.
+   */
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+
+  /**
+   * Cast this JSON value to a 16-bit signed integer.
+   *
+   * Calls get_int64() and checks that the result fits in an int16_t.
+   *
+   * @returns A 16-bit signed integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int16_t.
+   */
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+
+  /**
+   * Cast this JSON value to an 8-bit unsigned integer.
+   *
+   * Calls get_uint64() and checks that the result fits in a uint8_t.
+   *
+   * @returns An 8-bit unsigned integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint8_t.
+   */
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+
+  /**
+   * Cast this JSON value to an 8-bit signed integer.
+   *
+   * Calls get_int64() and checks that the result fits in an int8_t.
+   *
+   * @returns An 8-bit signed integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int8_t.
+   */
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
+
   /**
    * Cast this JSON value to a double.
    *
@@ -118334,6 +149667,53 @@ public:
    */
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;

+  /**
+   * Cast this JSON value to a float (binary32).
+   *
+   * The value is rounded directly to binary32: it is the float nearest to the
+   * JSON number. Note that this may differ from get_double() followed by a cast
+   * to float, which rounds twice.
+   *
+   * @returns A float.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+
+  /**
+   * Cast this JSON value (inside string) to a float (binary32).
+   *
+   * @returns A float.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  /**
+   * Cast this JSON value to a std::float32_t (C++23).
+   *
+   * Same as get_float(): the value is rounded directly to binary32.
+   *
+   * @returns A std::float32_t.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  /**
+   * Cast this JSON value to a std::float64_t (C++23).
+   *
+   * Same as get_double().
+   *
+   * @returns A std::float64_t.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   */
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
+
   /**
    * Cast this JSON value to a string.
    *
@@ -118361,6 +149741,26 @@ public:
    */
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;

+#if SIMDJSON_SUPPORTS_CHAR8_T
+  /**
+   * Cast this JSON value to a C++20 UTF-8 string.
+   *
+   * The string is guaranteed to be valid UTF-8.
+   *
+   * Equivalent to get<std::u8string_view>(). This is a zero-copy alias of
+   * get_string(): the very same bytes are returned, viewed as char8_t.
+   *
+   * Important: a value should be consumed once. Calling get_u8string() twice on the same
+   * value is an error.
+   *
+   * @param allow_replacement Whether to allow a replacement character for unmatched surrogate pairs.
+   * @returns An UTF-8 string. The string is stored in the parser and will be invalidated the next
+   *          time it parses a document or when it is destroyed.
+   * @returns INCORRECT_TYPE if the JSON value is not a string.
+   */
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
   /**
    * Attempts to fill the provided std::string reference with the parsed value of the current string.
    *
@@ -118448,7 +149848,7 @@ public:
   /**
    * Cast this JSON value to an unsigned integer.
    *
-   * @returns A signed 64-bit integer.
+   * @returns A unsigned 64-bit integer.
    * @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit unsigned integer.
    */
   simdjson_inline operator uint64_t() noexcept(false);
@@ -118913,9 +150313,24 @@ public:
   simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template <typename string_type>
   simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -118923,9 +150338,19 @@ public:
   simdjson_inline simdjson_result<bool> get_bool() noexcept;
   simdjson_inline simdjson_result<bool> is_null() noexcept;

-  template<typename T> simdjson_inline simdjson_result<T> get() noexcept;
+  template<typename T> simdjson_inline simdjson_result<T> get()
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, ppc64::ondemand::value>);
+#else
+    noexcept;
+#endif

-  template<typename T> simdjson_inline error_code get(T &out) noexcept;
+  template<typename T> simdjson_inline error_code get(T &out)
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, ppc64::ondemand::value>);
+#else
+    noexcept;
+#endif

 #if SIMDJSON_EXCEPTIONS
   template <class T>
@@ -119256,6 +150681,7 @@ protected:
   token_position _position{};

   friend class json_iterator;
+  friend class document_stream;
   friend class value_iterator;
   friend class object;
   template <typename... Args>
@@ -119347,6 +150773,9 @@ protected:
    * value of this attribute.
    */
   bool _streaming{false};
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  bool _allow_incomplete_json{false};
+#endif

 public:
   simdjson_inline json_iterator() noexcept = default;
@@ -119371,6 +150800,10 @@ public:
    * start_root_array() and start_root_object().
    */
   simdjson_inline bool streaming() const noexcept;
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  simdjson_inline bool allow_incomplete_json() const noexcept;
+  simdjson_inline size_t remaining_input_length(const uint8_t *json) const noexcept;
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON

   /**
    * Get the root value iterator
@@ -120250,33 +151683,87 @@ public:
    * @param batch_size The batch size to use. MUST be larger than the largest document. The sweet
    *                   spot is cache-related: small enough to fit in cache, yet big enough to
    *                   parse as many documents as possible in one tight loop.
-   *                   Defaults to 10MB, which has been a reasonable sweet spot in our tests.
-   * @param allow_comma_separated (defaults on false) This allows a mode where the documents are
-   *                   separated by commas instead of whitespace. It comes with a performance
-   *                   penalty because the entire document is indexed at once (and the document must be
-   *                   less than 4 GB), and there is no multithreading. In this mode, the batch_size parameter
-   *                   is effectively ignored, as it is set to at least the document size.
+   *                   Defaults to 1MB, which has been a reasonable sweet spot in our tests.
+   * @param allow_comma_separated @deprecated Use stream_format::comma_delimited instead.
+   *                   When true, maps internally to stream_format::comma_delimited.
+   *                   Defaults to false.
    * @return The stream, or an error. An empty input will yield 0 documents rather than an EMPTY error. Errors:
    *         - MEMALLOC if the parser does not have enough capacity and memory allocation fails
    *         - CAPACITY if the parser does not have enough capacity and batch_size > max_capacity.
    *         - other json errors if parsing fails. You should not rely on these errors to always the same for the
    *           same document: they may vary under runtime dispatch (so they may vary depending on your system and hardware).
    */
-  inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
-  inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
-  inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
-  inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size)
+  inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size)
     the string might be automatically padded with up to SIMDJSON_PADDING whitespace characters */
-  inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
-  inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
+  inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @private An rvalue input is destroyed at the end of the full-expression, while the
+   * returned document_stream only holds a pointer to it: iterating the stream would then
+   * read freed memory. These deleted overloads also catch a std::string_view argument,
+   * which would otherwise convert implicitly to a padded_string temporary. */
+  inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size = DEFAULT_BATCH_SIZE) = delete;// unsafe
+  /** @private @overload iterate_many(const std::string &&s, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size = DEFAULT_BATCH_SIZE) = delete;// unsafe
   /** @private We do not want to allow implicit conversion from C string to std::string. */
   simdjson_result<document_stream> iterate_many(const char *buf, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept = delete;

+#ifndef SIMDJSON_DISABLE_DEPRECATED_API
+  /**
+   * @deprecated Use iterate_many with stream_format::comma_delimited instead.
+   */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @private Deleted for the same reason as iterate_many(const std::string &&s, size_t batch_size). */
+  inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size, bool allow_comma_separated) = delete;// unsafe
+  /** @private @overload iterate_many(const std::string &&s, size_t batch_size, bool allow_comma_separated) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size, bool allow_comma_separated) = delete;// unsafe
+#endif // SIMDJSON_DISABLE_DEPRECATED_API
+
+  /**
+   * Parse a stream of JSON documents with explicit format specification.
+   *
+   * @param buf The concatenated JSON documents.
+   * @param len The length of the buffer.
+   * @param batch_size The batch size to use.
+   * @param format The stream format (whitespace_delimited, json_sequence, or comma_delimited).
+   * @return A stream of documents, or an error.
+   */
+  inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(padded_string_view s, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format)
+   *
+   * The string is padded in place if needed (see simdjson::pad), as with iterate_many(std::string &s, size_t batch_size).
+   */
+  inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size, stream_format format) noexcept;
+  /** @private Deleted for the same reason as iterate_many(const std::string &&s, size_t batch_size). */
+  inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+  /** @private @overload iterate_many(const std::string &&s, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+
   /** The capacity of this parser (the largest document it can process). */
   simdjson_pure simdjson_inline size_t capacity() const noexcept;
   /** The maximum capacity of this parser (the largest document it is allowed to process). */
@@ -120404,6 +151891,7 @@ private:
   size_t _capacity{0};
   size_t _max_capacity;
   size_t _max_depth{DEFAULT_MAX_DEPTH};
+  size_t _document_len{0};
   std::unique_ptr<uint8_t[]> string_buf{};

 #if SIMDJSON_DEVELOPMENT_CHECKS
@@ -120466,8 +151954,19 @@ public:
    * Begin array iteration.
    *
    * Part of the std::iterable interface.
+   *
+   * With SIMDJSON_DEVELOPMENT_CHECKS, the iterator locks this array while it is
+   * alive, so that reentrant access (at(), count_elements(), reset(), ...) is
+   * reported as OUT_OF_ORDER_ITERATION.
+   */
+  simdjson_inline simdjson_result<array_iterator> begin() & noexcept;
+  /**
+   * Begin iteration over a temporary array, e.g., `v.get_array().begin()`.
+   *
+   * The iterator does not depend on the array instance and may outlive it, so
+   * it does not lock it.
    */
-  simdjson_inline simdjson_result<array_iterator> begin() noexcept;
+  simdjson_inline simdjson_result<array_iterator> begin() && noexcept;
   /**
    * Sentinel representing the end of the array.
    *
@@ -120598,7 +152097,7 @@ public:
    */
   template <typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out)
-     noexcept(custom_deserializable<T, array> ? nothrow_custom_deserializable<T, array> : true) {
+     noexcept(nothrow_gettable<T, array>) {
     static_assert(custom_deserializable<T, array>);
     return deserialize(*this, out);
   }
@@ -120610,7 +152109,7 @@ public:
    */
   template <typename T>
   simdjson_inline simdjson_result<T> get()
-    noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+    noexcept(nothrow_gettable<T, array>)
   {
     static_assert(std::is_default_constructible<T>::value, "The specified type is not default constructible.");
     T out{};
@@ -120666,6 +152165,10 @@ protected:
    * iter.is_alive() == false indicates iteration is complete.
    */
   value_iterator iter{};
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  bool locked{false};
+  simdjson_inline void set_locked(bool _locked) noexcept;
+#endif

   friend class value;
   friend class document;
@@ -120687,7 +152190,8 @@ public:
   simdjson_inline simdjson_result(error_code error) noexcept; ///< @private
   simdjson_inline simdjson_result() noexcept = default;

-  simdjson_inline simdjson_result<ppc64::ondemand::array_iterator> begin() noexcept;
+  simdjson_inline simdjson_result<ppc64::ondemand::array_iterator> begin() & noexcept;
+  simdjson_inline simdjson_result<ppc64::ondemand::array_iterator> begin() && noexcept;
   simdjson_inline simdjson_result<ppc64::ondemand::array_iterator> end() noexcept;
   inline simdjson_result<size_t> count_elements() & noexcept;
   inline simdjson_result<bool> is_empty() & noexcept;
@@ -120707,7 +152211,7 @@ public:
   // TODO: move this code into object-inl.h

   template<typename T>
-  simdjson_inline simdjson_result<T> get() noexcept {
+  simdjson_inline simdjson_result<T> get() noexcept(nothrow_gettable<T, ppc64::ondemand::array>) {
     if (error()) { return error(); }
     if constexpr (std::is_same_v<T, ppc64::ondemand::array>) {
       return first;
@@ -120715,7 +152219,7 @@ public:
     return first.get<T>();
   }
   template<typename T>
-  simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept {
+  simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept(nothrow_gettable<T, ppc64::ondemand::array>) {
     if (error()) { return error(); }
     if constexpr (std::is_same_v<T, ppc64::ondemand::array>) {
       out = first;
@@ -120767,6 +152271,15 @@ public:
   /** Create a new, invalid array iterator. */
   simdjson_inline array_iterator() noexcept = default;

+#if SIMDJSON_DEVELOPMENT_CHECKS
+  simdjson_inline ~array_iterator() noexcept;
+
+  simdjson_inline array_iterator(array_iterator&&) noexcept;
+  simdjson_inline array_iterator& operator=(array_iterator&&) noexcept;
+  simdjson_inline array_iterator(const array_iterator&) noexcept;
+  simdjson_inline array_iterator& operator=(const array_iterator&) noexcept;
+#endif
+
   //
   // Iterator interface
   //
@@ -120809,6 +152322,9 @@ public:
 private:
 #if SIMDJSON_DEVELOPMENT_CHECKS
    bool has_been_referenced{false};
+   array* parent{nullptr};
+
+   simdjson_inline array_iterator(const value_iterator &_iter, array* _parent) noexcept;
 #endif
   value_iterator iter{};

@@ -120908,14 +152424,14 @@ public:
   /**
    * Cast this JSON value to an unsigned integer.
    *
-   * @returns A signed 64-bit integer.
+   * @returns A unsigned 64-bit integer.
    * @returns INCORRECT_TYPE If the JSON value is not a 64-bit unsigned integer.
    */
   simdjson_inline simdjson_result<uint64_t> get_uint64() noexcept;
   /**
    * Cast this JSON value (inside string) to an unsigned integer.
    *
-   * @returns A signed 64-bit integer.
+   * @returns A unsigned 64-bit integer.
    * @returns INCORRECT_TYPE If the JSON value is not a 64-bit unsigned integer.
    */
   simdjson_inline simdjson_result<uint64_t> get_uint64_in_string() noexcept;
@@ -120953,6 +152469,46 @@ public:
    * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int32_t.
    */
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  /**
+   * Cast this JSON value to a 16-bit unsigned integer.
+   *
+   * Calls get_uint64() and checks that the result fits in a uint16_t.
+   *
+   * @returns A 16-bit unsigned integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint16_t.
+   */
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  /**
+   * Cast this JSON value to a 16-bit signed integer.
+   *
+   * Calls get_int64() and checks that the result fits in an int16_t.
+   *
+   * @returns A 16-bit signed integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int16_t.
+   */
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  /**
+   * Cast this JSON value to an 8-bit unsigned integer.
+   *
+   * Calls get_uint64() and checks that the result fits in a uint8_t.
+   *
+   * @returns An 8-bit unsigned integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint8_t.
+   */
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  /**
+   * Cast this JSON value to an 8-bit signed integer.
+   *
+   * Calls get_int64() and checks that the result fits in an int8_t.
+   *
+   * @returns An 8-bit signed integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int8_t.
+   */
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   /**
    * Cast this JSON value to a double.
    *
@@ -120968,6 +152524,53 @@ public:
    * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
    */
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+
+  /**
+   * Cast this JSON value to a float (binary32).
+   *
+   * The value is rounded directly to binary32: it is the float nearest to the
+   * JSON number. Note that this may differ from get_double() followed by a cast
+   * to float, which rounds twice.
+   *
+   * @returns A float.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+
+  /**
+   * Cast this JSON value (inside string) to a float (binary32).
+   *
+   * @returns A float.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  /**
+   * Cast this JSON value to a std::float32_t (C++23).
+   *
+   * Same as get_float(): the value is rounded directly to binary32.
+   *
+   * @returns A std::float32_t.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  /**
+   * Cast this JSON value to a std::float64_t (C++23).
+   *
+   * Same as get_double().
+   *
+   * @returns A std::float64_t.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   */
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   /**
    * Cast this JSON value to a string.
    *
@@ -120981,6 +152584,24 @@ public:
    * @returns INCORRECT_TYPE if the JSON value is not a string.
    */
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  /**
+   * Cast this JSON value to a C++20 UTF-8 string.
+   *
+   * The string is guaranteed to be valid UTF-8.
+   *
+   * Equivalent to get<std::u8string_view>(). This is a zero-copy alias of
+   * get_string(): the very same bytes are returned, viewed as char8_t.
+   *
+   * Important: Calling get_u8string() twice on the same document is an error.
+   *
+   * @param allow_replacement Whether to allow a replacement character for unmatched surrogate pairs.
+   * @returns An UTF-8 string. The string is stored in the parser and will be invalidated the next
+   *          time it parses a document or when it is destroyed.
+   * @returns INCORRECT_TYPE if the JSON value is not a string.
+   */
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   /**
    * Attempts to fill the provided std::string reference with the parsed value of the current string.
    *
@@ -121051,9 +152672,12 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool
    *
-   * You may use get_double(), get_bool(), get_uint64(), get_int64(),
+   * You may use get_double(), get_float(), get_bool(), get_uint64(), get_int64(),
+   * get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(),
+   * get_float32(), get_float64(),
    * get_object(), get_array(), get_raw_json_string(), or get_string() instead.
    *
    * @returns A value of the given type, parsed from the JSON.
@@ -121062,7 +152686,7 @@ public:
   template <typename T>
   simdjson_inline simdjson_result<T> get() &
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    noexcept(nothrow_gettable<T, document>)
 #else
     noexcept
 #endif
@@ -121085,7 +152709,7 @@ public:
   template<typename T>
   simdjson_inline simdjson_result<T> get() &&
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    noexcept(nothrow_gettable<T, document>)
 #else
     noexcept
 #endif
@@ -121097,7 +152721,8 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool, value
+   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool, value
    *
    * Be mindful that the document instance must remain in scope while you are accessing object, array and value instances.
    *
@@ -121108,7 +152733,7 @@ public:
   template<typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out) &
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    noexcept(nothrow_gettable<T, document>)
 #else
     noexcept
 #endif
@@ -121121,7 +152746,7 @@ public:
         "And you do not seem to have added support for it. Indeed, we have that "
         "simdjson::custom_deserializable<T> is false and the type T is not a default type "
         "such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-        "int64_t, double, or bool.");
+        "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
       static_cast<void>(out); // to get rid of unused errors
       return UNINITIALIZED;
     }
@@ -121130,7 +152755,8 @@ public:
     // immediately fail.
     static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
       "The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-      "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+      "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+      "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
       " get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
       " You may also add support for custom types, see our documentation.");
     static_cast<void>(out); // to get rid of unused errors
@@ -121139,7 +152765,12 @@ public:
   }

   /** @overload template<typename T> error_code get(T &out) & noexcept */
-  template<typename T> simdjson_deprecated simdjson_inline error_code get(T &out) && noexcept;
+  template<typename T> simdjson_deprecated simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, document>);
+#else
+    noexcept;
+#endif

 #if SIMDJSON_EXCEPTIONS
   /**
@@ -121173,24 +152804,24 @@ public:
   /**
    * Cast this JSON value to an unsigned integer.
    *
-   * @returns A signed 64-bit integer.
+   * @returns A unsigned 64-bit integer.
    * @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit unsigned integer.
    */
-  simdjson_inline operator uint64_t() noexcept(false);
+  explicit simdjson_inline operator uint64_t() noexcept(false);
   /**
    * Cast this JSON value to a signed integer.
    *
    * @returns A signed 64-bit integer.
    * @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit integer.
    */
-  simdjson_inline operator int64_t() noexcept(false);
+  explicit simdjson_inline operator int64_t() noexcept(false);
   /**
    * Cast this JSON value to a double.
    *
    * @returns A double.
    * @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a valid floating-point number.
    */
-  simdjson_inline operator double() noexcept(false);
+  explicit simdjson_inline operator double() noexcept(false);
   /**
    * Cast this JSON value to a string.
    *
@@ -121200,7 +152831,7 @@ public:
    *          time it parses a document or when it is destroyed.
    * @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not a string.
    */
-  simdjson_inline operator std::string_view() noexcept(false) simdjson_lifetime_bound;
+  explicit simdjson_inline operator std::string_view() noexcept(false) simdjson_lifetime_bound;
   /**
    * Cast this JSON value to a raw_json_string.
    *
@@ -121209,14 +152840,14 @@ public:
    * @returns A pointer to the raw JSON for the given string.
    * @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not a string.
    */
-  simdjson_inline operator raw_json_string() noexcept(false) simdjson_lifetime_bound;
+  explicit simdjson_inline operator raw_json_string() noexcept(false) simdjson_lifetime_bound;
   /**
    * Cast this JSON value to a bool.
    *
    * @returns A bool value.
    * @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not true or false.
    */
-  simdjson_inline operator bool() noexcept(false);
+  explicit simdjson_inline operator bool() noexcept(false);
   /**
    * Cast this JSON value to a value when the document is an object or an array.
    *
@@ -121711,9 +153342,24 @@ public:
   simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template <typename string_type>
   simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -121725,7 +153371,7 @@ public:
   template <typename T>
   simdjson_inline simdjson_result<T> get() &
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    noexcept(nothrow_gettable<T, document_reference>)
 #else
     noexcept
 #endif
@@ -121738,7 +153384,8 @@ public:
   template<typename T>
   simdjson_inline simdjson_result<T> get() &&
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    // Forwards to document::get<T>(), so the document customization decides.
+    noexcept(nothrow_gettable<T, document>)
 #else
     noexcept
 #endif
@@ -121750,7 +153397,8 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool, value
+   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool, value
    *
    * Be mindful that the document instance must remain in scope while you are accessing object, array and value instances.
    *
@@ -121761,7 +153409,7 @@ public:
   template<typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out) &
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document_reference> : true)
+    noexcept(nothrow_gettable<T, document_reference>)
 #else
     noexcept
 #endif
@@ -121774,7 +153422,7 @@ public:
         "And you do not seem to have added support for it. Indeed, we have that "
         "simdjson::custom_deserializable<T> is false and the type T is not a default type "
         "such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-        "int64_t, double, or bool.");
+        "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
       static_cast<void>(out); // to get rid of unused errors
       return UNINITIALIZED;
     }
@@ -121783,7 +153431,8 @@ public:
     // immediately fail.
     static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
       "The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-      "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+      "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+      "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
       " get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
       " You may also add support for custom types, see our documentation.");
     static_cast<void>(out); // to get rid of unused errors
@@ -121792,7 +153441,12 @@ public:
   }

   /** @overload template<typename T> error_code get(T &out) & noexcept */
-  template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+  template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, document_reference>);
+#else
+    noexcept;
+#endif
   simdjson_inline simdjson_result<std::string_view> raw_json() noexcept;
 #if SIMDJSON_STATIC_REFLECTION
   template<constevalutil::fixed_string... FieldNames, typename T>
@@ -121805,12 +153459,12 @@ public:
   explicit simdjson_inline operator T() noexcept(false);
   simdjson_inline operator array() & noexcept(false);
   simdjson_inline operator object() & noexcept(false);
-  simdjson_inline operator uint64_t() noexcept(false);
-  simdjson_inline operator int64_t() noexcept(false);
-  simdjson_inline operator double() noexcept(false);
-  simdjson_inline operator std::string_view() noexcept(false);
-  simdjson_inline operator raw_json_string() noexcept(false);
-  simdjson_inline operator bool() noexcept(false);
+  explicit simdjson_inline operator uint64_t() noexcept(false);
+  explicit simdjson_inline operator int64_t() noexcept(false);
+  explicit simdjson_inline operator double() noexcept(false);
+  explicit simdjson_inline operator std::string_view() noexcept(false);
+  explicit simdjson_inline operator raw_json_string() noexcept(false);
+  explicit simdjson_inline operator bool() noexcept(false);
   simdjson_inline operator value() noexcept(false);
 #endif
   simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -121872,9 +153526,24 @@ public:
   simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template <typename string_type>
   simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -121883,11 +153552,31 @@ public:
   simdjson_inline simdjson_result<ppc64::ondemand::value> get_value() noexcept;
   simdjson_inline simdjson_result<bool> is_null() noexcept;

-  template<typename T> simdjson_inline simdjson_result<T> get() & noexcept;
-  template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() && noexcept;
+  template<typename T> simdjson_inline simdjson_result<T> get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, ppc64::ondemand::document>);
+#else
+    noexcept;
+#endif
+  template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, ppc64::ondemand::document>);
+#else
+    noexcept;
+#endif

-  template<typename T> simdjson_inline error_code get(T &out) & noexcept;
-  template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+  template<typename T> simdjson_inline error_code get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, ppc64::ondemand::document>);
+#else
+    noexcept;
+#endif
+  template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, ppc64::ondemand::document>);
+#else
+    noexcept;
+#endif
 #if SIMDJSON_EXCEPTIONS

   using ppc64::implementation_simdjson_result_base<ppc64::ondemand::document>::operator*;
@@ -121896,12 +153585,12 @@ public:
   explicit simdjson_inline operator T() noexcept(false);
   simdjson_inline operator ppc64::ondemand::array() & noexcept(false);
   simdjson_inline operator ppc64::ondemand::object() & noexcept(false);
-  simdjson_inline operator uint64_t() noexcept(false);
-  simdjson_inline operator int64_t() noexcept(false);
-  simdjson_inline operator double() noexcept(false);
-  simdjson_inline operator std::string_view() noexcept(false);
-  simdjson_inline operator ppc64::ondemand::raw_json_string() noexcept(false);
-  simdjson_inline operator bool() noexcept(false);
+  explicit simdjson_inline operator uint64_t() noexcept(false);
+  explicit simdjson_inline operator int64_t() noexcept(false);
+  explicit simdjson_inline operator double() noexcept(false);
+  explicit simdjson_inline operator std::string_view() noexcept(false);
+  explicit simdjson_inline operator ppc64::ondemand::raw_json_string() noexcept(false);
+  explicit simdjson_inline operator bool() noexcept(false);
   simdjson_inline operator ppc64::ondemand::value() noexcept(false);
 #endif
   simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -121967,9 +153656,24 @@ public:
   simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template <typename string_type>
   simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -121978,22 +153682,42 @@ public:
   simdjson_inline simdjson_result<ppc64::ondemand::value> get_value() noexcept;
   simdjson_inline simdjson_result<bool> is_null() noexcept;

-  template<typename T> simdjson_inline simdjson_result<T> get() & noexcept;
-  template<typename T> simdjson_inline simdjson_result<T> get() && noexcept;
+  template<typename T> simdjson_inline simdjson_result<T> get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, ppc64::ondemand::document_reference>);
+#else
+    noexcept;
+#endif
+  template<typename T> simdjson_inline simdjson_result<T> get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, ppc64::ondemand::document_reference>);
+#else
+    noexcept;
+#endif

-  template<typename T> simdjson_inline error_code get(T &out) & noexcept;
-  template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+  template<typename T> simdjson_inline error_code get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, ppc64::ondemand::document_reference>);
+#else
+    noexcept;
+#endif
+  template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, ppc64::ondemand::document_reference>);
+#else
+    noexcept;
+#endif
 #if SIMDJSON_EXCEPTIONS
   template <class T>
   explicit simdjson_inline operator T() noexcept(false);
   simdjson_inline operator ppc64::ondemand::array() & noexcept(false);
   simdjson_inline operator ppc64::ondemand::object() & noexcept(false);
-  simdjson_inline operator uint64_t() noexcept(false);
-  simdjson_inline operator int64_t() noexcept(false);
-  simdjson_inline operator double() noexcept(false);
-  simdjson_inline operator std::string_view() noexcept(false);
-  simdjson_inline operator ppc64::ondemand::raw_json_string() noexcept(false);
-  simdjson_inline operator bool() noexcept(false);
+  explicit simdjson_inline operator uint64_t() noexcept(false);
+  explicit simdjson_inline operator int64_t() noexcept(false);
+  explicit simdjson_inline operator double() noexcept(false);
+  explicit simdjson_inline operator std::string_view() noexcept(false);
+  explicit simdjson_inline operator ppc64::ondemand::raw_json_string() noexcept(false);
+  explicit simdjson_inline operator bool() noexcept(false);
   simdjson_inline operator ppc64::ondemand::value() noexcept(false);
 #endif
   simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -122161,10 +153885,7 @@ public:
    *   }
    *   size_t truncated = stream.truncated_bytes();
    *
-   * IMPORTANT: this value is only meaningful under the conditions below. It is
-   * computed from stage-1 bookkeeping, and outside these conditions it is not
-   * merely imprecise, it is arbitrary -- it can exceed size_in_bytes() or wrap
-   * around to a huge value. Check it only when both of the following hold:
+   * IMPORTANT: this value is only meaningful under the conditions below.
    *
    *   - you iterated all the way to the end of the stream;
    *   - no document reported an error. Iteration stops at the first failed
@@ -122173,6 +153894,9 @@ public:
    * If you need to know about a truncated tail outside those conditions, track
    * it yourself from the last successful document (see iterator::current_index()
    * and iterator::source()).
+   *
+   * An empty input (zero bytes) or an input made only of white space contains
+   * no document: truncated_bytes() returns zero.
    */
   inline size_t truncated_bytes() const noexcept;

@@ -122232,7 +153956,10 @@ public:
      *
      * The returned string_view instance is simply a map to the (unparsed)
      * source string: it may thus include white-space characters and all manner
-     * of padding.
+     * of padding. It spans the whole current document, whether or not you
+     * have already accessed (part of) the document. Thus
+     * current_index() + source().size() is the offset just past the end of the
+     * current document, which is useful when reading a stream in chunks.
      *
      * This function (source()) is experimental and the usage
      * may change in future versions of simdjson: we find the API somewhat
@@ -122286,13 +154013,16 @@ private:
    * @param buf is the raw byte buffer we need to process
    * @param len is the length of the raw byte buffer in bytes
    * @param batch_size is the size of the windows (must be strictly greater or equal to the largest JSON document)
+   * @param allow_comma_separated whether to allow comma-separated documents
+   * @param format the stream format
    */
   simdjson_inline document_stream(
     ondemand::parser &parser,
     const uint8_t *buf,
     size_t len,
     size_t batch_size,
-    bool allow_comma_separated
+    bool allow_comma_separated,
+    stream_format format = stream_format::whitespace_delimited
   ) noexcept;

   /**
@@ -122326,8 +154056,23 @@ private:
    */
   inline void next() noexcept;

-  /** Move the json_iterator of the document to the location of the next document in the stream. */
+  /**
+   * Move the json_iterator of the document to the location of the next document
+   * in the stream.
+   *
+   * For formats with a document delimiter (`newline_delimited`, `json_sequence`),
+   * when the iterator is still inside the current document (`depth() > 0`), this
+   * may jump to the next delimiter instead of walking remaining structurals. That
+   * jump does not structure-validate the unread remainder.
+   */
   inline void next_document() noexcept;
+  /** Byte that ends a document under `format`, or 0 if there is none. */
+  simdjson_inline uint8_t document_delimiter() const noexcept;
+  /**
+   * Position the iterator at the first structural at or past the next
+   * `delimiter` in the current batch. Returns false if none is found.
+   */
+  simdjson_inline bool skip_to_delimiter(uint8_t delimiter) noexcept;

   /** Get the next document index. */
   inline size_t next_batch_start() const noexcept;
@@ -122341,6 +154086,7 @@ private:
   size_t len;
   size_t batch_size;
   bool allow_comma_separated;
+  stream_format format;
   /**
    * We are going to use just one document instance. The document owns
    * the json_iterator. It implies that we only ever pass a reference
@@ -122367,7 +154113,7 @@ private:
   /** The error returned from the stage 1 thread. */
   error_code stage1_thread_error{UNINITIALIZED};
   /** The thread used to run stage 1 against the next batch in the background. */
-  std::unique_ptr<stage1_worker> worker{new(std::nothrow) stage1_worker()};
+  std::unique_ptr<stage1_worker> worker{};
   /**
    * The parser used to run stage 1 in the background. Will be swapped
    * with the regular parser when finished.
@@ -122442,6 +154188,16 @@ public:
    * call it again nor can you call key().
    */
   simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  /**
+   * Get the key as a C++20 u8string_view. This is a zero-copy alias of
+   * unescaped_key(): the very same bytes are returned, viewed as char8_t.
+   *
+   * This consumes the key: once you have called unescaped_u8key(), you cannot
+   * call it again nor can you call key().
+   */
+  simdjson_inline simdjson_warn_unused simdjson_result<std::u8string_view> unescaped_u8key(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   /**
    * Get the key as a string_view (for higher speed, consider raw_key).
    * We deliberately use a more cumbersome name (unescaped_key) to force users
@@ -122474,6 +154230,16 @@ public:
    * you can safely call it repeatedly. Use unescaped_key() to get the unescaped key.
    */
   simdjson_inline std::string_view escaped_key() const noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  /**
+   * Get the key as a C++20 u8string_view. This is a zero-copy alias of
+   * escaped_key(): the very same bytes are returned, viewed as char8_t.
+   * The string is unprocessed, so it may contain escape characters
+   * (e.g., \uXXXX or \n). It does not count as a consumption of the content:
+   * you can safely call it repeatedly.
+   */
+  simdjson_inline std::u8string_view escaped_u8key() const noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   /**
    * Get the field value.
    */
@@ -122505,11 +154271,17 @@ public:
   simdjson_inline simdjson_result() noexcept = default;

   simdjson_inline simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> unescaped_u8key(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template<typename string_type>
   simdjson_inline error_code unescaped_key(string_type &receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<ppc64::ondemand::raw_json_string> key() noexcept;
   simdjson_inline simdjson_result<std::string_view> key_raw_json_token() noexcept;
   simdjson_inline simdjson_result<std::string_view> escaped_key() noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> escaped_u8key() noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   simdjson_inline simdjson_result<ppc64::ondemand::value> value() noexcept;
 };

@@ -122517,6 +154289,1398 @@ public:

 #endif // SIMDJSON_GENERIC_ONDEMAND_FIELD_H
 /* end file simdjson/generic/ondemand/field.h for ppc64 */
+/* including simdjson/generic/ondemand/key_selector.h for ppc64: #include "simdjson/generic/ondemand/key_selector.h" */
+/* begin file simdjson/generic/ondemand/key_selector.h for ppc64 */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+#define SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #include "simdjson/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/common_defs.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/constevalutil.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/raw_json_string.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#include <array>
+#include <string>      // std::string (key_selector::describe)
+#include <string_view>
+#include <cstddef>
+#include <cstdint>
+#include <cstring>     // std::memcpy (portable unaligned window load)
+#include <utility>     // std::index_sequence (window candidate dispatch)
+#include <type_traits> // std::integral_constant (window candidate dispatch)
+
+// AArch64 only: 32-bit ARM (ARMv7) defines __ARM_NEON too, but lacks the
+// A64-only horizontal reductions (vmaxvq_u32) used below. See issue #2885.
+#if defined(__aarch64__)
+  #include <arm_neon.h>
+  #define SIMDJSON_KEY_SELECTOR_HAS_NEON 1
+#else
+  #define SIMDJSON_KEY_SELECTOR_HAS_NEON 0
+#endif
+#if defined(__SSE2__)
+  #include <emmintrin.h>
+  #define SIMDJSON_KEY_SELECTOR_HAS_SSE2 1
+#else
+  #define SIMDJSON_KEY_SELECTOR_HAS_SSE2 0
+#endif
+#if defined(__loongarch_sx)
+  #include <lsxintrin.h>
+  #define SIMDJSON_KEY_SELECTOR_HAS_LSX 1
+#else
+  #define SIMDJSON_KEY_SELECTOR_HAS_LSX 0
+#endif
+
+#if SIMDJSON_SUPPORTS_CONCEPTS
+
+namespace simdjson {
+namespace ppc64 {
+namespace ondemand {
+
+namespace key_selector_detail {
+
+// Since not constexpr, triggers compile-time error.
+inline void compile_time_error(const char* message) noexcept { (void)message; }
+
+// ============================================================================
+// Compile-time perfect-hash generator.
+// It scales to ~100 keys at compile time by determining association values one (position, character)
+// symbol at a time (gperf-style) instead of an exhaustive offset search, and
+// falls back to a Hash-and-Displace construction for large/awkward key sets.
+//
+// Only flat tables survive to runtime; the lookup is a few additions plus a
+// single SIMD key comparison (see match_raw below).
+// ============================================================================
+
+// Maximum number of character positions the gperf hash may combine.
+static constexpr std::size_t MAX_POSITIONS = 16;
+// Sentinel "position" meaning "the last character of the key".
+static constexpr std::size_t LAST_CHAR = std::size_t(-1);
+// Runtime-encoded sentinels (stored in uint8 tables).
+static constexpr std::uint8_t POS_LAST_CHAR = 0xFF; // positions_[i] == last char
+static constexpr std::uint8_t HD_MODE       = 0xFF; // num_positions == H&D mode
+// Flags stored in positions[2] in H&D mode to select the key-hash variant.
+static constexpr std::size_t HD_HASH_2BYTE_FLAG = 2;
+static constexpr std::size_t HD_HASH_4BYTE_FLAG = 4;
+
+constexpr std::size_t next_power_of_2(std::size_t n) noexcept {
+    if (n == 0) { return 1; }
+    std::size_t p = 1;
+    while (p < n) { p <<= 1; }
+    return p;
+}
+
+// Character at a given position (LAST_CHAR means last character), or 256 if out
+// of bounds.
+constexpr std::size_t char_at(std::string_view key, std::size_t pos) noexcept {
+    if (pos == LAST_CHAR) {
+        if (key.empty()) { return 256; }
+        return static_cast<unsigned char>(key[key.size() - 1]);
+    }
+    if (pos >= key.size()) { return 256; }
+    return static_cast<unsigned char>(key[pos]);
+}
+
+// Count key pairs that a set of positions fails to distinguish. Keys whose
+// lengths differ modulo the table size are separated by the length term in the
+// hash, so they need no position coverage.
+template <std::size_t N>
+consteval std::size_t count_undistinguished_pairs(
+    const std::array<std::string_view, N>& keys,
+    const std::size_t* positions,
+    std::size_t num_positions,
+    std::size_t modulus) {
+    std::size_t count = 0;
+    for (std::size_t i = 0; i < N; ++i) {
+        for (std::size_t j = i + 1; j < N; ++j) {
+            if (keys[i].size() % modulus != keys[j].size() % modulus) { continue; }
+            bool distinguished = false;
+            for (std::size_t p = 0; p < num_positions; ++p) {
+                if (char_at(keys[i], positions[p]) != char_at(keys[j], positions[p])) {
+                    distinguished = true;
+                    break;
+                }
+            }
+            if (!distinguished) { ++count; }
+        }
+    }
+    return count;
+}
+
+template <std::size_t N>
+consteval bool positions_distinguish(
+    const std::array<std::string_view, N>& keys,
+    const std::size_t* positions,
+    std::size_t num_positions,
+    std::size_t modulus) {
+    return count_undistinguished_pairs<N>(keys, positions, num_positions, modulus) == 0;
+}
+
+// Number of distinct (length % modulus, char_at(key, pos)) pairs at a position.
+template <std::size_t N>
+consteval std::size_t discriminating_power(
+    const std::array<std::string_view, N>& keys,
+    std::size_t pos,
+    std::size_t modulus) {
+    struct pair { std::size_t len_mod; std::size_t ch; };
+    std::array<pair, N> pairs{};
+    for (std::size_t i = 0; i < N; ++i) {
+        pairs[i] = {keys[i].size() % modulus, char_at(keys[i], pos)};
+    }
+    std::size_t count = 0;
+    for (std::size_t i = 0; i < N; ++i) {
+        bool dup = false;
+        for (std::size_t j = 0; j < i; ++j) {
+            if (pairs[i].len_mod == pairs[j].len_mod && pairs[i].ch == pairs[j].ch) {
+                dup = true;
+                break;
+            }
+        }
+        if (!dup) { ++count; }
+    }
+    return count;
+}
+
+template <std::size_t N>
+consteval std::size_t max_key_length(const std::array<std::string_view, N>& keys) {
+    std::size_t m = 0;
+    for (std::size_t i = 0; i < N; ++i) {
+        if (keys[i].size() > m) { m = keys[i].size(); }
+    }
+    return m;
+}
+
+// Bounded backtracking DFS for a minimal set of distinguishing positions.
+template <std::size_t N>
+consteval bool backtracking_search(
+    const std::array<std::string_view, N>& keys,
+    const std::size_t* candidates,
+    std::size_t num_candidates,
+    std::size_t* positions,
+    std::size_t& num_positions_out,
+    std::size_t& budget,
+    std::size_t modulus) {
+    constexpr std::size_t MAX_DEPTH = 8;
+    std::size_t breadth = num_candidates < 20 ? num_candidates : 20;
+
+    struct frame { std::size_t depth; std::size_t next_ci; std::size_t parent_count; };
+    std::array<frame, MAX_DEPTH + 1> stack{};
+    std::size_t sp = 0;
+
+    std::size_t initial_count = count_undistinguished_pairs<N>(keys, positions, 0, modulus);
+    if (budget > 0) { --budget; }
+    if (initial_count == 0) { num_positions_out = 0; return true; }
+
+    stack[0] = {0, 0, initial_count};
+
+    while (budget > 0) {
+        if (sp > MAX_DEPTH) {
+            if (sp == 0) { break; }
+            --sp;
+            ++stack[sp].next_ci;
+            continue;
+        }
+        auto& f = stack[sp];
+        if (f.next_ci >= breadth) {
+            if (sp == 0) { break; }
+            --sp;
+            ++stack[sp].next_ci;
+            continue;
+        }
+        positions[sp] = candidates[f.next_ci];
+        --budget;
+        std::size_t new_count = count_undistinguished_pairs<N>(keys, positions, sp + 1, modulus);
+        if (new_count == 0) { num_positions_out = sp + 1; return true; }
+        if (new_count < f.parent_count && sp + 1 < MAX_DEPTH) {
+            stack[sp + 1] = {sp + 1, f.next_ci + 1, new_count};
+            ++sp;
+        } else {
+            ++f.next_ci;
+        }
+    }
+    return false;
+}
+
+// Phase 1: select character positions that distinguish all colliding pairs.
+template <std::size_t N>
+consteval std::size_t select_positions(
+    const std::array<std::string_view, N>& keys,
+    std::array<std::size_t, MAX_POSITIONS>& positions,
+    std::size_t modulus) {
+    if (positions_distinguish<N>(keys, positions.data(), 0, modulus)) { return 0; }
+
+    std::size_t max_len = max_key_length(keys);
+    constexpr std::size_t MAX_CANDIDATES = 256;
+    std::array<std::size_t, MAX_CANDIDATES> candidates{};
+    std::array<std::size_t, MAX_CANDIDATES> powers{};
+    std::size_t num_candidates = 0;
+    for (std::size_t p = 0; p < max_len && num_candidates < MAX_CANDIDATES - 1; ++p) {
+        candidates[num_candidates] = p;
+        powers[num_candidates] = discriminating_power(keys, p, modulus);
+        ++num_candidates;
+    }
+    if (num_candidates < MAX_CANDIDATES) {
+        candidates[num_candidates] = LAST_CHAR;
+        powers[num_candidates] = discriminating_power(keys, LAST_CHAR, modulus);
+        ++num_candidates;
+    }
+    for (std::size_t i = 0; i < num_candidates; ++i) {
+        for (std::size_t j = i + 1; j < num_candidates; ++j) {
+            if (powers[j] > powers[i]) {
+                auto tc = candidates[i]; candidates[i] = candidates[j]; candidates[j] = tc;
+                auto tp = powers[i]; powers[i] = powers[j]; powers[j] = tp;
+            }
+        }
+    }
+
+    positions[0] = candidates[0];
+    if (positions_distinguish<N>(keys, positions.data(), 1, modulus)) { return 1; }
+
+    positions[0] = 0;
+    positions[1] = LAST_CHAR;
+    if (positions_distinguish<N>(keys, positions.data(), 2, modulus)) { return 2; }
+
+    {
+        std::size_t budget = 5000;
+        std::size_t num_found = 0;
+        if (backtracking_search<N>(keys, candidates.data(), num_candidates,
+                                   positions.data(), num_found, budget, modulus)) {
+            return num_found;
+        }
+    }
+
+    std::size_t num_pos = 0;
+    for (std::size_t ci = 0; ci < num_candidates && num_pos < MAX_POSITIONS; ++ci) {
+        bool already = false;
+        for (std::size_t p = 0; p < num_pos; ++p) {
+            if (positions[p] == candidates[ci]) { already = true; break; }
+        }
+        if (already) { continue; }
+        positions[num_pos] = candidates[ci];
+        ++num_pos;
+        if (positions_distinguish<N>(keys, positions.data(), num_pos, modulus)) { return num_pos; }
+    }
+
+    compile_time_error("Failed to find distinguishing positions for perfect hash");
+    return 0;
+}
+
+// Result of PHF computation. A max-sized slot_to_key array lets the same struct
+// type carry any chosen table size.
+template <std::size_t N>
+struct phf_result {
+    // Allow up to 8x the minimum table size. Sparser tables solve faster.
+    static constexpr std::size_t MAX_TABLE_SIZE = next_power_of_2(N) * 8;
+    std::size_t table_size{};
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso_values{};
+    std::size_t num_positions{};
+    std::array<std::size_t, MAX_POSITIONS> positions{};
+    std::array<std::size_t, MAX_TABLE_SIZE> slot_to_key{};
+};
+
+// Partition-based asso_values search (gperf-style). Determines asso_values one
+// (position, character) symbol at a time; never revisits a value. Equivalence
+// classes (keys sharing the same undetermined symbols) keep the search cheap.
+template <std::size_t N, std::size_t M>
+consteval bool try_generate_gperf(
+    const std::array<std::string_view, N>& keys,
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS>& asso_values,
+    std::size_t& num_positions,
+    std::array<std::size_t, MAX_POSITIONS>& positions,
+    std::array<std::size_t, M>& slot_to_key) {
+    num_positions = select_positions<N>(keys, positions, M);
+
+    for (std::size_t p = 0; p < MAX_POSITIONS; ++p) {
+        for (std::size_t c = 0; c < 256; ++c) { asso_values[p][c] = 0; }
+    }
+
+    if (num_positions == 0) {
+        for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+        for (std::size_t i = 0; i < N; ++i) {
+            std::size_t slot = keys[i].size() % M;
+            if (slot_to_key[slot] != N) { return false; }
+            slot_to_key[slot] = i;
+        }
+        return true;
+    }
+
+    std::array<std::array<std::size_t, MAX_POSITIONS>, N> kchars{};
+    for (std::size_t k = 0; k < N; ++k) {
+        for (std::size_t p = 0; p < num_positions; ++p) {
+            kchars[k][p] = char_at(keys[k], positions[p]);
+        }
+    }
+
+    struct sym_t { std::size_t pos; std::size_t ch; std::size_t freq; };
+    constexpr std::size_t MAX_SYMS = MAX_POSITIONS * 256;
+    std::array<sym_t, MAX_SYMS> syms{};
+    std::size_t nsyms = 0;
+    for (std::size_t p = 0; p < num_positions; ++p) {
+        std::array<std::size_t, 256> freq{};
+        for (std::size_t k = 0; k < N; ++k) {
+            std::size_t c = kchars[k][p];
+            if (c < 256) { freq[c]++; }
+        }
+        for (std::size_t c = 0; c < 256; ++c) {
+            if (freq[c] > 0) { syms[nsyms++] = {p, c, freq[c]}; }
+        }
+    }
+    for (std::size_t i = 0; i < nsyms; ++i) {
+        for (std::size_t j = i + 1; j < nsyms; ++j) {
+            if (syms[j].freq > syms[i].freq) {
+                auto tmp = syms[i]; syms[i] = syms[j]; syms[j] = tmp;
+            }
+        }
+    }
+
+    std::array<std::size_t, N> phash{};
+    for (std::size_t k = 0; k < N; ++k) { phash[k] = keys[k].size(); }
+
+    std::array<std::array<uint64_t, 256>, MAX_POSITIONS> salt{};
+    {
+        uint64_t s = 0x9e3779b97f4a7c15ULL;
+        for (std::size_t p = 0; p < num_positions; ++p) {
+            for (std::size_t c = 0; c < 256; ++c) {
+                s = s * 6364136223846793005ULL + 1442695040888963407ULL;
+                salt[p][c] = s;
+            }
+        }
+    }
+    std::array<uint64_t, N> sig{};
+    for (std::size_t k = 0; k < N; ++k) {
+        uint64_t s = 0;
+        for (std::size_t p = 0; p < num_positions; ++p) {
+            std::size_t c = kchars[k][p];
+            if (c < 256) { s ^= salt[p][c]; }
+        }
+        sig[k] = s;
+    }
+    std::array<std::size_t, N> order{};
+    for (std::size_t k = 0; k < N; ++k) { order[k] = k; }
+
+    std::array<std::size_t, M> slot_gen{};
+    std::size_t gen = 0;
+
+    std::size_t search_limit = next_power_of_2(M);
+    if (search_limit < 32) { search_limit = 32; }
+
+    for (std::size_t si = 0; si < nsyms; ++si) {
+        std::size_t sp = syms[si].pos;
+        std::size_t sc = syms[si].ch;
+
+        uint64_t sp_salt = salt[sp][sc];
+        for (std::size_t k = 0; k < N; ++k) {
+            if (kchars[k][sp] == sc) { sig[k] ^= sp_salt; }
+        }
+
+        for (std::size_t i = 1; i < N; ++i) {
+            std::size_t x = order[i];
+            uint64_t xs = sig[x];
+            std::size_t j = i;
+            while (j > 0 && sig[order[j - 1]] > xs) {
+                order[j] = order[j - 1];
+                --j;
+            }
+            order[j] = x;
+        }
+
+        bool found = false;
+        for (std::size_t v = 0; v < search_limit && !found; ++v) {
+            bool collision = false;
+            std::size_t ci = 0;
+            while (ci < N && !collision) {
+                uint64_t class_sig = sig[order[ci]];
+                std::size_t cj = ci;
+                while (cj < N && sig[order[cj]] == class_sig) { ++cj; }
+                if (cj - ci > 1) {
+                    ++gen;
+                    for (std::size_t x = ci; x < cj; ++x) {
+                        std::size_t k = order[x];
+                        std::size_t h = phash[k];
+                        if (kchars[k][sp] == sc) { h += v; }
+                        h %= M;
+                        if (slot_gen[h] == gen) { collision = true; break; }
+                        slot_gen[h] = gen;
+                    }
+                }
+                ci = cj;
+            }
+            if (!collision) {
+                asso_values[sp][sc] = v;
+                for (std::size_t k = 0; k < N; ++k) {
+                    if (kchars[k][sp] == sc) { phash[k] += v; }
+                }
+                found = true;
+            }
+        }
+        if (!found) { return false; }
+    }
+
+    for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+    for (std::size_t i = 0; i < N; ++i) {
+        std::size_t slot = phash[i] % M;
+        if (slot_to_key[slot] != N) { return false; }
+        slot_to_key[slot] = i;
+    }
+    std::size_t filled = 0;
+    for (std::size_t i = 0; i < M; ++i) {
+        if (slot_to_key[i] != N) { ++filled; }
+    }
+    return filled == N;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_compute_phf(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+    static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso{};
+    std::size_t npos{};
+    std::array<std::size_t, MAX_POSITIONS> pos{};
+    std::array<std::size_t, M> s2k{};
+    if (try_generate_gperf<N, M>(keys, asso, npos, pos, s2k)) {
+        result.table_size = M;
+        result.asso_values = asso;
+        result.num_positions = npos;
+        result.positions = pos;
+        for (std::size_t i = 0; i < M; ++i) { result.slot_to_key[i] = s2k[i]; }
+        for (std::size_t i = M; i < phf_result<N>::MAX_TABLE_SIZE; ++i) { result.slot_to_key[i] = N; }
+        return true;
+    }
+    return false;
+}
+
+template <std::size_t N, std::size_t M, std::size_t MaxM>
+consteval bool try_gperf_po2(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+    if (try_compute_phf<N, M>(keys, result)) { return true; }
+    constexpr std::size_t NextM = M * 2;
+    if constexpr (NextM <= MaxM) { return try_gperf_po2<N, NextM, MaxM>(keys, result); }
+    return false;
+}
+
+// --- Hash-and-Displace fallback --------------------------------------------
+
+constexpr std::size_t hd_bucket_hash(std::string_view key) noexcept {
+    std::size_t c0 = key.empty() ? 0 : static_cast<unsigned char>(key[0]);
+    std::size_t c1 = key.empty() ? 0 : static_cast<unsigned char>(key[key.size() - 1]);
+    return (c0 + c1 * 3 + key.size() * 17) & 0xFF;
+}
+constexpr std::size_t hd_safe_char(const char* p, std::size_t len, std::size_t idx) noexcept {
+    std::size_t has = static_cast<std::size_t>(idx < len);
+    std::size_t si = idx & (std::size_t{0} - has);
+    return static_cast<unsigned char>(p[si]) & (std::size_t{0} - has);
+}
+constexpr std::size_t hd_key_hash_2(std::string_view key) noexcept {
+    std::size_t kc = key.size();
+    kc = kc * 31 + static_cast<unsigned char>(key[0]);
+    kc = kc * 31 + hd_safe_char(key.data(), key.size(), 1);
+    return kc;
+}
+constexpr std::size_t hd_key_hash_4(std::string_view key) noexcept {
+    std::size_t kc = key.size();
+    kc = kc * 31 + static_cast<unsigned char>(key[0]);
+    kc = kc * 31 + hd_safe_char(key.data(), key.size(), 1);
+    kc = kc * 31 + hd_safe_char(key.data(), key.size(), 2);
+    kc = kc * 31 + hd_safe_char(key.data(), key.size(), 3);
+    return kc;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_hash_and_displace(
+    const std::array<std::string_view, N>& keys,
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS>& asso_values,
+    std::size_t& num_positions,
+    std::array<std::size_t, MAX_POSITIONS>& positions,
+    std::array<std::size_t, M>& slot_to_key) {
+    num_positions = select_positions<N>(keys, positions, M);
+
+    if (num_positions == 0) {
+        for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+        for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+        for (std::size_t i = 0; i < N; ++i) {
+            std::size_t slot = keys[i].size() % M;
+            if (slot_to_key[slot] != N) { return false; }
+            slot_to_key[slot] = i;
+        }
+        return true;
+    }
+
+    for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+    num_positions = HD_MODE; // sentinel for H&D mode
+    positions[0] = 0;
+    positions[1] = LAST_CHAR;
+
+    std::array<std::size_t, N> key_bucket{};
+    for (std::size_t i = 0; i < N; ++i) { key_bucket[i] = hd_bucket_hash(keys[i]); }
+
+    struct bucket_info { std::size_t ch; std::size_t count; };
+    std::array<bucket_info, N> buckets{};
+    std::size_t num_buckets = 0;
+    for (std::size_t i = 0; i < N; ++i) {
+        std::size_t bk = key_bucket[i];
+        bool found = false;
+        for (std::size_t b = 0; b < num_buckets; ++b) {
+            if (buckets[b].ch == bk) { ++buckets[b].count; found = true; break; }
+        }
+        if (!found) { buckets[num_buckets++] = {bk, 1}; }
+    }
+    for (std::size_t i = 0; i < num_buckets; ++i) {
+        for (std::size_t j = i + 1; j < num_buckets; ++j) {
+            if (buckets[j].count > buckets[i].count) {
+                auto tmp = buckets[i]; buckets[i] = buckets[j]; buckets[j] = tmp;
+            }
+        }
+    }
+
+    auto try_placement = [&](auto key_hash_fn) -> bool {
+        for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+        for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+        for (std::size_t b = 0; b < num_buckets; ++b) {
+            std::size_t ch = buckets[b].ch;
+            std::array<std::size_t, N> bucket_keys{};
+            std::size_t bk_count = 0;
+            for (std::size_t i = 0; i < N; ++i) {
+                if (key_bucket[i] == ch) { bucket_keys[bk_count++] = i; }
+            }
+            bool placed = false;
+            std::size_t max_d = M < 255 ? M : 255;
+            for (std::size_t d = 0; d < max_d; ++d) {
+                bool ok = true;
+                std::array<std::size_t, N> bucket_slots{};
+                for (std::size_t k = 0; k < bk_count; ++k) {
+                    std::size_t slot = (d + key_hash_fn(keys[bucket_keys[k]])) % M;
+                    if (slot_to_key[slot] != N) { ok = false; break; }
+                    for (std::size_t k2 = 0; k2 < k; ++k2) {
+                        if (bucket_slots[k2] == slot) { ok = false; break; }
+                    }
+                    if (!ok) { break; }
+                    bucket_slots[k] = slot;
+                }
+                if (ok) {
+                    asso_values[0][ch] = d;
+                    for (std::size_t k = 0; k < bk_count; ++k) {
+                        slot_to_key[bucket_slots[k]] = bucket_keys[k];
+                    }
+                    placed = true;
+                    break;
+                }
+            }
+            if (!placed) { return false; }
+        }
+        std::size_t filled = 0;
+        for (std::size_t i = 0; i < M; ++i) {
+            if (slot_to_key[i] != N) { ++filled; }
+        }
+        return filled == N;
+    };
+
+    if (try_placement([](std::string_view k) { return hd_key_hash_2(k); })) {
+        positions[2] = HD_HASH_2BYTE_FLAG;
+        return true;
+    }
+    if (try_placement([](std::string_view k) { return hd_key_hash_4(k); })) {
+        positions[2] = HD_HASH_4BYTE_FLAG;
+        return true;
+    }
+    return false;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_compute_phf_hd(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+    static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso{};
+    std::size_t npos{};
+    std::array<std::size_t, MAX_POSITIONS> pos{};
+    std::array<std::size_t, M> s2k{};
+    if (try_hash_and_displace<N, M>(keys, asso, npos, pos, s2k)) {
+        result.table_size = M;
+        result.asso_values = asso;
+        result.num_positions = npos;
+        result.positions = pos;
+        for (std::size_t i = 0; i < M; ++i) { result.slot_to_key[i] = s2k[i]; }
+        for (std::size_t i = M; i < phf_result<N>::MAX_TABLE_SIZE; ++i) { result.slot_to_key[i] = N; }
+        return true;
+    }
+    return false;
+}
+
+template <std::size_t N, std::size_t M>
+consteval phf_result<N> compute_phf_hd_po2(const std::array<std::string_view, N>& keys) {
+    static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+    phf_result<N> result{};
+    if (try_compute_phf_hd<N, M>(keys, result)) { return result; }
+    constexpr std::size_t NextM = M * 2;
+    if constexpr (NextM <= phf_result<N>::MAX_TABLE_SIZE) {
+        return compute_phf_hd_po2<N, NextM>(keys);
+    } else {
+        compile_time_error("Hash-and-Displace: failed to find valid table size");
+        return result;
+    }
+}
+
+// Compute a perfect hash for `keys`: try gperf at power-of-two sizes (capped so
+// the runtime tables stay within uint8 indices), then fall back to H&D.
+template <std::size_t N>
+consteval phf_result<N> compute_phf(const std::array<std::string_view, N>& keys) {
+    constexpr std::size_t StartM = next_power_of_2(N);
+    constexpr std::size_t GPERF_MAX_TABLE =
+        phf_result<N>::MAX_TABLE_SIZE < 256 ? phf_result<N>::MAX_TABLE_SIZE : 256;
+    if constexpr (StartM <= GPERF_MAX_TABLE) {
+        phf_result<N> result{};
+        if (try_gperf_po2<N, StartM, GPERF_MAX_TABLE>(keys, result)) { return result; }
+    }
+    return compute_phf_hd_po2<N, StartM>(keys);
+}
+
+// ============================================================================
+// Runtime tables (flat, uint8) derived from a phf_result.
+// ============================================================================
+
+template <std::size_t N, std::size_t TableSize, std::size_t MaxKeyLen>
+struct phf_data {
+    std::array<std::array<std::uint8_t, 256>, MAX_POSITIONS> asso_values{};
+    std::array<std::uint8_t, MAX_POSITIONS>                  positions{};
+    std::uint8_t                                             num_positions{};
+    std::uint8_t                                             hd_hash_variant{}; // 2 or 4 (H&D only)
+    std::array<std::uint8_t, TableSize>                      slot_to_key{};
+    // slot_key_bytes[s] holds the key stored at slot s, zero-padded to a 16-byte
+    // multiple so the SIMD comparison can read a whole register.
+    std::array<std::array<char, ((MaxKeyLen + 15) / 16) * 16>, TableSize> slot_key_bytes{};
+    std::array<std::uint8_t, TableSize>                      slot_key_len{};
+};
+
+template <std::size_t N>
+constexpr std::size_t compute_max_key_len(const std::array<std::string_view, N>& keys) noexcept {
+    std::size_t m = 0;
+    for (std::size_t i = 0; i < N; ++i) { if (keys[i].size() > m) { m = keys[i].size(); } }
+    return m;
+}
+
+// Validate keys and build the runtime tables from the computed perfect hash.
+template <std::size_t N, std::size_t TableSize, std::size_t MaxKeyLen>
+consteval phf_data<N, TableSize, MaxKeyLen>
+build_phf_data(const std::array<std::string_view, N>& keys, const phf_result<N>& result) {
+    for (std::size_t i = 0; i < N; ++i) {
+        if (keys[i].empty())            { compile_time_error("empty keys are not allowed in key_selector"); }
+        if (keys[i].size() > MaxKeyLen) { compile_time_error("key length exceeds MaxKeyLen"); }
+        for (char c : keys[i]) {
+            if (c == '\\') { compile_time_error("backslash not allowed in key_selector keys"); }
+            if (c == '"')  { compile_time_error("quote not allowed in key_selector keys"); }
+            if (c == '\0') { compile_time_error("null byte not allowed in key_selector keys"); }
+        }
+        for (std::size_t j = i + 1; j < N; ++j) {
+            if (keys[i] == keys[j]) { compile_time_error("duplicate keys in key_selector"); }
+        }
+    }
+
+    phf_data<N, TableSize, MaxKeyLen> out{};
+
+    if (result.num_positions == HD_MODE) {
+        // H&D mode: single displacement table in asso_values[0].
+        for (std::size_t c = 0; c < 256; ++c) {
+            out.asso_values[0][c] = static_cast<std::uint8_t>(result.asso_values[0][c]);
+        }
+        out.num_positions   = static_cast<std::uint8_t>(HD_MODE);
+        out.hd_hash_variant = static_cast<std::uint8_t>(result.positions[2]);
+    } else {
+        for (std::size_t pi = 0; pi < result.num_positions; ++pi) {
+            for (std::size_t c = 0; c < 256; ++c) {
+                out.asso_values[pi][c] = static_cast<std::uint8_t>(result.asso_values[pi][c] % TableSize);
+            }
+        }
+        out.num_positions = static_cast<std::uint8_t>(result.num_positions);
+        for (std::size_t i = 0; i < result.num_positions; ++i) {
+            out.positions[i] = (result.positions[i] == LAST_CHAR)
+                ? POS_LAST_CHAR
+                : static_cast<std::uint8_t>(result.positions[i]);
+        }
+    }
+
+    for (std::size_t s = 0; s < TableSize; ++s) {
+        out.slot_to_key[s] = static_cast<std::uint8_t>(result.slot_to_key[s]);
+    }
+
+    for (std::size_t s = 0; s < TableSize; ++s) {
+        std::size_t ki = result.slot_to_key[s];
+        if (ki < N) {
+            auto k = keys[ki];
+            out.slot_key_len[s] = static_cast<std::uint8_t>(k.size());
+            for (std::size_t c = 0; c < k.size(); ++c) { out.slot_key_bytes[s][c] = k[c]; }
+        } else {
+            out.slot_key_len[s] = 0; // empty slot: no length can match
+        }
+    }
+    return out;
+}
+
+// --- SIMD runtime primitives ------------------------------------------------
+
+// The runtime matchers read whole 16-byte blocks up to offset 63 (four blocks)
+// for keys as long as the 63-character maximum. That read must stay within the
+// buffer's trailing padding, so the padding has to exceed the largest offset we
+// touch.
+static_assert(SIMDJSON_PADDING > 63,
+              "key_selector requires SIMDJSON_PADDING > 63 for its SIMD key reads");
+
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+// True when every byte of `diff` is zero. A 32-bit-lane horizontal max is
+// enough (zero iff every 32-bit word is zero) and is cheaper than a byte-wide
+// reduction. Do not replace this with a floating-point compare against 0.0:
+// flush-to-zero would treat a denormal as zero.
+simdjson_really_inline bool neon_all_bytes_zero(uint8x16_t diff) noexcept {
+    return vmaxvq_u32(vreinterpretq_u32_u8(diff)) == 0;
+}
+#endif
+
+// Scan for the terminating '"' starting at p. Returns its byte offset (= key
+// length). Caller guarantees SIMDJSON_PADDING (== 64) bytes past the JSON buffer,
+// so reading whole 16-byte blocks up to offset 63 is always safe.
+template <std::size_t MaxKeyLen>
+simdjson_really_inline std::size_t scan_key_length(const char* p) noexcept {
+    // The closing quote of the longest key sits at most at offset MaxKeyLen, so we
+    // scan as many 16-byte blocks as it takes to cover offsets 0..MaxKeyLen. The
+    // cap (63) keeps every covered offset within the 64-byte padding guarantee, so
+    // the SIMD and scalar builds agree.
+    static_assert(MaxKeyLen <= 63, "MaxKeyLen must be <= 63 for current SIMD implementations");
+    // [[maybe_unused]]: the scalar fallback (no NEON/SSE2/LSX) does not use it.
+    [[maybe_unused]] constexpr std::size_t num_blocks = MaxKeyLen / 16 + 1;
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+    for (std::size_t b = 0; b < num_blocks; ++b) {
+        uint8x16_t v = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + b * 16);
+        uint8x16_t cmp = vceqq_u8(v, vdupq_n_u8('"'));
+        uint64_t m = vget_lane_u64(
+            vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(cmp), 4)), 0);
+        if (simdjson_likely(m != 0)) { return b * 16 + (std::size_t(__builtin_ctzll(m)) >> 2); }
+    }
+    return MaxKeyLen + 1;
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+    for (std::size_t b = 0; b < num_blocks; ++b) {
+        __m128i v   = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + b * 16));
+        __m128i cmp = _mm_cmpeq_epi8(v, _mm_set1_epi8('"'));
+        unsigned m  = static_cast<unsigned>(_mm_movemask_epi8(cmp));
+        if (simdjson_likely(m != 0)) { return b * 16 + std::size_t(__builtin_ctz(m)); }
+    }
+    return MaxKeyLen + 1;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+    for (std::size_t b = 0; b < num_blocks; ++b) {
+        __m128i v   = __lsx_vld(reinterpret_cast<const void*>(p + b * 16), 0);
+        __m128i cmp = __lsx_vseq_b(v, __lsx_vreplgr2vr_b('"'));
+        // vmskltz_b gathers the per-byte sign bits (set where the byte equals '"')
+        // into the low 16 bits of lane 0, the LSX equivalent of movemask.
+        unsigned m  = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(cmp), 0)) & 0xFFFFu;
+        if (simdjson_likely(m != 0)) { return b * 16 + std::size_t(__builtin_ctz(m)); }
+    }
+    return MaxKeyLen + 1;
+#else
+    for (std::size_t i = 0; i <= MaxKeyLen; ++i)
+        if (p[i] == '"') return i;
+    return MaxKeyLen + 1;
+#endif
+}
+
+// Byte-equal of p[0..len) against stored[0..len). stored is zero-padded past
+// `len`. Input is read over 16, 32, 48, or 64 bytes (padded JSON buffer
+// guaranteed; SIMDJSON_PADDING == 64).
+template <std::size_t MaxKeyLen>
+simdjson_really_inline bool compare_key_bytes(
+    const char* p, const char* stored, std::size_t len) noexcept {
+    // [[maybe_unused]]: only the NEON/SSE2/LSX branches read these; the scalar
+    // fallback build (no SIMD) leaves them unused, which is an error under -Werror.
+    [[maybe_unused]] alignas(16) static constexpr uint8_t idx16[16] =
+        {0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15};
+    if constexpr (MaxKeyLen <= 16) {
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+        uint8x16_t vp = vld1q_u8(reinterpret_cast<const uint8_t*>(p));
+        uint8x16_t vs = vld1q_u8(reinterpret_cast<const uint8_t*>(stored));
+        uint8x16_t mask = vcltq_u8(vld1q_u8(idx16), vdupq_n_u8(static_cast<uint8_t>(len)));
+        uint8x16_t diff = veorq_u8(vandq_u8(vp, mask), vs);
+        return neon_all_bytes_zero(diff);
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+        __m128i vp = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p));
+        __m128i vs = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored));
+        __m128i idx = _mm_load_si128(reinterpret_cast<const __m128i*>(idx16));
+        __m128i mask = _mm_cmplt_epi8(idx, _mm_set1_epi8(static_cast<char>(len)));
+        __m128i eq = _mm_cmpeq_epi8(_mm_and_si128(vp, mask), vs);
+        return _mm_movemask_epi8(eq) == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+        __m128i vp = __lsx_vld(reinterpret_cast<const void*>(p), 0);
+        __m128i vs = __lsx_vld(reinterpret_cast<const void*>(stored), 0);
+        __m128i idx = __lsx_vld(reinterpret_cast<const void*>(idx16), 0);
+        __m128i mask = __lsx_vslt_b(idx, __lsx_vreplgr2vr_b(static_cast<int>(len)));
+        __m128i eq = __lsx_vseq_b(__lsx_vand_v(vp, mask), vs);
+        return (static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq), 0)) & 0xFFFFu) == 0xFFFFu;
+#else
+        for (std::size_t i = 0; i < len; ++i)
+            if (p[i] != stored[i]) return false;
+        return true;
+#endif
+    } else if constexpr (MaxKeyLen <= 32) {
+        [[maybe_unused]] alignas(16) static constexpr uint8_t idx32_hi[16] =
+            {16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31};
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+        uint8x16_t vp_lo = vld1q_u8(reinterpret_cast<const uint8_t*>(p));
+        uint8x16_t vp_hi = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + 16);
+        uint8x16_t vs_lo = vld1q_u8(reinterpret_cast<const uint8_t*>(stored));
+        uint8x16_t vs_hi = vld1q_u8(reinterpret_cast<const uint8_t*>(stored) + 16);
+        uint8x16_t lenv = vdupq_n_u8(static_cast<uint8_t>(len));
+        uint8x16_t m_lo = vcltq_u8(vld1q_u8(idx16),    lenv);
+        uint8x16_t m_hi = vcltq_u8(vld1q_u8(idx32_hi), lenv);
+        uint8x16_t d_lo = veorq_u8(vandq_u8(vp_lo, m_lo), vs_lo);
+        uint8x16_t d_hi = veorq_u8(vandq_u8(vp_hi, m_hi), vs_hi);
+        return neon_all_bytes_zero(vorrq_u8(d_lo, d_hi));
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+        __m128i vp_lo = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p));
+        __m128i vp_hi = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + 16));
+        __m128i vs_lo = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored));
+        __m128i vs_hi = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored + 16));
+        __m128i lenv = _mm_set1_epi8(static_cast<char>(len));
+        __m128i m_lo = _mm_cmplt_epi8(_mm_load_si128(reinterpret_cast<const __m128i*>(idx16)),    lenv);
+        __m128i m_hi = _mm_cmplt_epi8(_mm_load_si128(reinterpret_cast<const __m128i*>(idx32_hi)), lenv);
+        __m128i eq_lo = _mm_cmpeq_epi8(_mm_and_si128(vp_lo, m_lo), vs_lo);
+        __m128i eq_hi = _mm_cmpeq_epi8(_mm_and_si128(vp_hi, m_hi), vs_hi);
+        return (_mm_movemask_epi8(eq_lo) & _mm_movemask_epi8(eq_hi)) == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+        __m128i lenv = __lsx_vreplgr2vr_b(static_cast<int>(len));
+        __m128i vp_lo = __lsx_vld(reinterpret_cast<const void*>(p), 0);
+        __m128i vp_hi = __lsx_vld(reinterpret_cast<const void*>(p + 16), 0);
+        __m128i vs_lo = __lsx_vld(reinterpret_cast<const void*>(stored), 0);
+        __m128i vs_hi = __lsx_vld(reinterpret_cast<const void*>(stored + 16), 0);
+        __m128i m_lo = __lsx_vslt_b(__lsx_vld(reinterpret_cast<const void*>(idx16), 0),    lenv);
+        __m128i m_hi = __lsx_vslt_b(__lsx_vld(reinterpret_cast<const void*>(idx32_hi), 0), lenv);
+        __m128i eq_lo = __lsx_vseq_b(__lsx_vand_v(vp_lo, m_lo), vs_lo);
+        __m128i eq_hi = __lsx_vseq_b(__lsx_vand_v(vp_hi, m_hi), vs_hi);
+        unsigned mlo = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq_lo), 0)) & 0xFFFFu;
+        unsigned mhi = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq_hi), 0)) & 0xFFFFu;
+        return (mlo & mhi) == 0xFFFFu;
+#else
+        for (std::size_t i = 0; i < len; ++i)
+            if (p[i] != stored[i]) return false;
+        return true;
+#endif
+    } else if constexpr (MaxKeyLen <= 64) {
+        // 3 or 4 16-byte blocks (33..48 -> 3, 49..64 -> 4). stored is zero-padded
+        // to exactly num_blocks*16 bytes (KEY_STRIDE), so neither load overruns it.
+        // [[maybe_unused]]: the scalar fallback (no NEON/SSE2/LSX) does not use it.
+        [[maybe_unused]] constexpr std::size_t num_blocks = (MaxKeyLen + 15) / 16;
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+        uint8x16_t base = vld1q_u8(idx16);
+        uint8x16_t lenv = vdupq_n_u8(static_cast<uint8_t>(len));
+        uint8x16_t acc  = vdupq_n_u8(0);
+        for (std::size_t b = 0; b < num_blocks; ++b) {
+            uint8x16_t vp   = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + b * 16);
+            uint8x16_t vs   = vld1q_u8(reinterpret_cast<const uint8_t*>(stored) + b * 16);
+            uint8x16_t idxv = vaddq_u8(base, vdupq_n_u8(static_cast<uint8_t>(b * 16)));
+            uint8x16_t mask = vcltq_u8(idxv, lenv);
+            acc = vorrq_u8(acc, veorq_u8(vandq_u8(vp, mask), vs));
+        }
+        return neon_all_bytes_zero(acc);
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+        __m128i base = _mm_load_si128(reinterpret_cast<const __m128i*>(idx16));
+        __m128i lenv = _mm_set1_epi8(static_cast<char>(len));
+        int eq = 0xFFFF;
+        for (std::size_t b = 0; b < num_blocks; ++b) {
+            __m128i vp   = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + b * 16));
+            __m128i vs   = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored + b * 16));
+            __m128i idxv = _mm_add_epi8(base, _mm_set1_epi8(static_cast<char>(b * 16)));
+            __m128i mask = _mm_cmplt_epi8(idxv, lenv);
+            eq &= _mm_movemask_epi8(_mm_cmpeq_epi8(_mm_and_si128(vp, mask), vs));
+        }
+        return eq == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+        __m128i base = __lsx_vld(reinterpret_cast<const void*>(idx16), 0);
+        __m128i lenv = __lsx_vreplgr2vr_b(static_cast<int>(len));
+        unsigned acc = 0xFFFFu;
+        for (std::size_t b = 0; b < num_blocks; ++b) {
+            __m128i vp   = __lsx_vld(reinterpret_cast<const void*>(p + b * 16), 0);
+            __m128i vs   = __lsx_vld(reinterpret_cast<const void*>(stored + b * 16), 0);
+            __m128i idxv = __lsx_vadd_b(base, __lsx_vreplgr2vr_b(static_cast<int>(b * 16)));
+            __m128i mask = __lsx_vslt_b(idxv, lenv);
+            __m128i eq   = __lsx_vseq_b(__lsx_vand_v(vp, mask), vs);
+            acc &= static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq), 0)) & 0xFFFFu;
+        }
+        return acc == 0xFFFFu;
+#else
+        for (std::size_t i = 0; i < len; ++i)
+            if (p[i] != stored[i]) return false;
+        return true;
+#endif
+    } else {
+        for (std::size_t i = 0; i < len; ++i)
+            if (p[i] != stored[i]) return false;
+        return true;
+    }
+}
+
+// --- Single 8-bit window fast path ------------------------------------------
+//
+// Many small key sets can be told apart by inspecting a *single* 8-bit window of
+// the key bytes -- and that window need not be byte-aligned. Because every JSON
+// key is terminated by a '"', the bytes at and before a key's length are well
+// defined for any key at least that long: byte i is the key character when i is
+// inside the key and the closing quote when i == len. So we read two bytes at a
+// fixed offset, extract 8 consecutive bits at a fixed intra-byte shift, and if
+// that value is distinct for every key, a 256-entry table maps it straight to a
+// candidate key. The match then needs no hash and -- in the length-free overload
+// -- no SIMD length scan: load two bytes, shift, mask, index the table, and
+// confirm the candidate with one comparison.
+//
+// Allowing an *unaligned* window (a shift of 1..7) mixes bits from two adjacent
+// bytes and discriminates key sets that no single aligned byte can. For example,
+// the partial_tweets keys {created_at,id,text,in_reply_to_status_id,user,
+// retweet_count,favorite_count} share a colliding byte at every aligned position
+// 0,1,2, yet the 8 bits starting at bit offset 2 are unique across all seven.
+// Simpler cases fall out as the shift==0 special case: {"id","screen_name"}
+// splits on byte 0, {"jo","joe"} on the quote at byte 2.
+//
+// The window is confined to the first (shortest key length + 1) bytes so the
+// two-byte read never crosses a key's closing quote into uncontrolled value
+// bytes; that final byte is the shortest key's quote.
+template <std::size_t N, std::size_t MaxKeyLen>
+struct window_data {
+    static constexpr std::size_t KEY_STRIDE = ((MaxKeyLen + 15) / 16) * 16;
+    bool                                            ok{false};
+    std::uint8_t                                    byte_offset{0}; // first byte of the 2-byte read
+    std::uint8_t                                    shift{0};       // intra-byte bit shift (0..7)
+    std::array<std::uint8_t, 256>                   window_to_key{}; // window byte -> key index, N if none
+    std::array<std::uint8_t, N>                     key_len{};
+    std::array<std::array<char, KEY_STRIDE>, N>     key_bytes{};     // zero-padded
+};
+
+// Byte seen at `idx` for key `i`: a key character when idx is inside the key, or
+// the closing '"' when idx == len (callers keep idx <= every key's length).
+template <std::size_t N>
+constexpr unsigned window_byte_at(const std::array<std::string_view, N>& keys,
+                                  std::size_t i, std::size_t idx) noexcept {
+    if (idx < keys[i].size()) { return static_cast<unsigned char>(keys[i][idx]); }
+    return static_cast<unsigned>('"');
+}
+
+// The 8-bit window value for key `i` at (byte_offset, shift): the little-endian
+// pair (byte[off], byte[off+1]) shifted right by `shift` and truncated. Matches
+// the runtime read exactly.
+template <std::size_t N>
+constexpr unsigned window_value(const std::array<std::string_view, N>& keys,
+                                std::size_t i, std::size_t byte_offset, std::size_t shift) noexcept {
+    unsigned lo = window_byte_at<N>(keys, i, byte_offset);
+    unsigned hi = window_byte_at<N>(keys, i, byte_offset + 1);
+    return ((lo | (hi << 8)) >> shift) & 0xFFu;
+}
+
+template <std::size_t N, std::size_t MaxKeyLen>
+consteval window_data<N, MaxKeyLen>
+compute_window(const std::array<std::string_view, N>& keys) {
+    window_data<N, MaxKeyLen> out{};
+
+    std::size_t min_len = keys[0].size();
+    for (std::size_t i = 1; i < N; ++i) {
+        if (keys[i].size() < min_len) { min_len = keys[i].size(); }
+    }
+
+    // Iterate windows nearest the front first (cheapest to read, smallest shift).
+    for (std::size_t off = 0; off <= min_len; ++off) {
+        for (std::size_t shift = 0; shift < 8; ++shift) {
+            // The read touches byte off, and byte off+1 when shift != 0. Both must
+            // stay within the safe region [0, min_len] (min_len is the shortest
+            // key's quote index). off <= min_len is guaranteed by the loop bound.
+            if (shift != 0 && off + 1 > min_len) { continue; }
+
+            bool distinct = true;
+            for (std::size_t i = 0; i < N && distinct; ++i) {
+                for (std::size_t j = i + 1; j < N; ++j) {
+                    if (window_value<N>(keys, i, off, shift) == window_value<N>(keys, j, off, shift)) {
+                        distinct = false;
+                        break;
+                    }
+                }
+            }
+            if (!distinct) { continue; }
+
+            out.ok          = true;
+            out.byte_offset = static_cast<std::uint8_t>(off);
+            out.shift       = static_cast<std::uint8_t>(shift);
+            for (std::size_t b = 0; b < 256; ++b) { out.window_to_key[b] = static_cast<std::uint8_t>(N); }
+            for (std::size_t i = 0; i < N; ++i) {
+                out.window_to_key[window_value<N>(keys, i, off, shift)] = static_cast<std::uint8_t>(i);
+                out.key_len[i] = static_cast<std::uint8_t>(keys[i].size());
+                for (std::size_t c = 0; c < keys[i].size(); ++c) { out.key_bytes[i][c] = keys[i][c]; }
+            }
+            return out;
+        }
+    }
+    return out; // ok == false: no single 8-bit window distinguishes the keys
+}
+
+// Read the 8-bit window at (byte_offset, shift) from a padded key pointer.
+simdjson_really_inline std::uint8_t read_window(const char* p, std::size_t byte_offset,
+                                                std::size_t shift) noexcept {
+    std::uint16_t w;
+    // Two controlled bytes (within the shortest key + its quote, hence within the
+    // padded buffer). memcpy is the portable little-endian unaligned load.
+    std::memcpy(&w, p + byte_offset, sizeof(w));
+#if defined(__BYTE_ORDER__) && (__BYTE_ORDER__ == __ORDER_BIG_ENDIAN__)
+    w = static_cast<std::uint16_t>((w >> 8) | (w << 8));
+#endif
+    return static_cast<std::uint8_t>((w >> shift) & 0xFFu);
+}
+
+// Verify a window candidate. The window table already mapped the key to the
+// single possible index `ki` (< N); here we confirm it. Folding over 0..N-1
+// turns the runtime `ki` into a compile-time index in the matching arm, so the
+// candidate's length and bytes are constants for compare_key_bytes -- the same
+// specialization the ordered find_field path gets from a CT-length
+// unsafe_is_equal, and what keeps this path competitive. The closing-quote check
+// (p[L] == '"') both confirms the key ends exactly at the candidate's length and
+// rejects a wrong-length key, so this works whether or not the caller knew len.
+template <std::size_t N, std::size_t MaxKeyLen, std::size_t... Is>
+simdjson_really_inline std::size_t
+match_window_candidate(const char* p, std::uint8_t ki,
+                       const window_data<N, MaxKeyLen>& w,
+                       std::index_sequence<Is...>) noexcept {
+  std::size_t result = N;
+  auto try_match = [&](auto Ic) {
+    constexpr std::size_t i = decltype(Ic)::value;
+    if (ki == i && p[w.key_len[i]] == '"' &&
+        key_selector_detail::compare_key_bytes<MaxKeyLen>(
+            p, w.key_bytes[i].data(), w.key_len[i])) {
+      result = i;
+    }
+  };
+  (try_match(std::integral_constant<std::size_t, Is>{}), ...);
+  return result;
+}
+
+// --- describe() string helpers (constexpr; no std::to_string, which is not) ---
+
+// Append the decimal form of v to s.
+SIMDJSON_CONSTEXPR_STRING void append_uint(std::string& s, std::size_t v) {
+    if (v == 0) { s.push_back('0'); return; }
+    char buf[20];
+    std::size_t n = 0;
+    while (v > 0) { buf[n++] = static_cast<char>('0' + (v % 10)); v /= 10; }
+    while (n > 0) { s.push_back(buf[--n]); }
+}
+
+// Append a byte as its decimal value, plus the printable character in quotes
+// when it is in the printable ASCII range (e.g. "110 ('n')").
+SIMDJSON_CONSTEXPR_STRING void append_byte(std::string& s, unsigned b) {
+    append_uint(s, b);
+    if (b >= 0x20 && b < 0x7f) {
+        s += " ('";
+        s.push_back(static_cast<char>(b));
+        s += "')";
+    }
+}
+
+} // namespace key_selector_detail
+
+/**
+ * Stateless, compile-time key selector.
+ *
+ * Usage:
+ *   using sel_t = key_selector<"id", "text", "user">;
+ *   std::size_t i = sel_t::match_raw(raw_key); // returns sel_t::size() on miss
+ *
+ * The perfect hash is built at compile time (gperf-style, with a
+ * Hash-and-Displace fallback) and only flat tables survive to runtime. All
+ * tables are static constexpr, so the lookup fully inlines.
+ *
+ * Limitations:
+ *   - Each key must be at most 63 characters long (and no longer than
+ *     SIMDJSON_PADDING). Longer keys trigger a compile-time error.
+ *   - The number of keys should be moderate. The hard limit is 255 keys;
+ *     compilation time grows with the number of keys, so prefer a few dozen at
+ *     most per selector.
+ *   - Keys must be distinct, non-empty, and free of backslash, double-quote and
+ *     null bytes (matching is done against the raw, unescaped JSON key bytes).
+ */
+template <constevalutil::fixed_string... Keys>
+struct key_selector {
+    static constexpr std::size_t N = sizeof...(Keys);
+    static_assert(N > 0,   "key_selector requires at least one key");
+    static_assert(N <= 255,"key_selector supports at most 255 keys");
+
+    static constexpr std::array<std::string_view, N> keys{ Keys.view()... };
+    static constexpr std::size_t max_key_len = key_selector_detail::compute_max_key_len<N>(keys);
+    static_assert(max_key_len <= SIMDJSON_PADDING,
+                  "key longer than SIMDJSON_PADDING is not supported");
+    // The SIMD key-length scan covers offsets 0..63 (four 16-byte blocks), which
+    // stays within the 64-byte padding guarantee. A 64-character key's closing
+    // quote would land at offset 64 and be missed on NEON/SSE2/LSX while still
+    // matching in scalar builds, so cap at 63 to keep implementations in agreement.
+    static_assert(max_key_len <= 63,
+                  "key_selector keys must be at most 63 characters long");
+
+    static constexpr auto result = key_selector_detail::compute_phf<N>(keys);
+    static constexpr std::size_t table_size = result.table_size;
+
+    static constexpr auto phf =
+        key_selector_detail::build_phf_data<N, table_size, max_key_len>(keys, result);
+
+    // Single 8-bit-window discriminator (when one exists). Detected at compile
+    // time and selected with `if constexpr` below, so the hash path is compiled
+    // out for key sets that qualify, and this is compiled out for those that do
+    // not.
+    static constexpr auto window =
+        key_selector_detail::compute_window<N, max_key_len>(keys);
+
+    static constexpr std::size_t size() noexcept { return N; }
+
+    /**
+     * Look up a JSON key whose length is already known. p must point at the first
+     * key byte (just after the opening quote) in a padded simdjson buffer, and len
+     * must be the number of raw key bytes (the distance to the closing quote).
+     * Returns the selector index in [0, N) on match, or N on miss.
+     *
+     * Prefer this overload when the caller can obtain the key length cheaply (for
+     * example, object::for_each derives it from the structural index rather than
+     * re-scanning for the closing quote).
+     */
+    static simdjson_really_inline std::size_t match_raw(const char* p, std::size_t len) noexcept {
+        if (len == 0 || len > max_key_len) { return N; }
+
+        if constexpr (window.ok) {
+            // One 8-bit window selects the only possible candidate key;
+            // match_window_candidate confirms it (bytes + closing quote). p sits
+            // in a padded buffer and the window stays within the shortest key +
+            // quote, so the two-byte read is always in bounds. len is unused here
+            // because the quote check already pins the key's end.
+            std::uint8_t ki = window.window_to_key[
+                key_selector_detail::read_window(p, window.byte_offset, window.shift)];
+            if (ki >= N) { return N; }
+            return key_selector_detail::match_window_candidate(
+                p, ki, window, std::make_index_sequence<N>{});
+        }
+
+        std::size_t slot;
+        if (phf.num_positions == key_selector_detail::HD_MODE) {
+            // Hash-and-Displace: bucket displacement + per-key hash.
+            std::string_view key(p, len);
+            std::size_t bucket = key_selector_detail::hd_bucket_hash(key);
+            std::size_t kh = (phf.hd_hash_variant == key_selector_detail::HD_HASH_2BYTE_FLAG)
+                ? key_selector_detail::hd_key_hash_2(key)
+                : key_selector_detail::hd_key_hash_4(key);
+            slot = (phf.asso_values[0][bucket] + kh) & (table_size - 1);
+        } else {
+            // gperf: h = len + sum of asso_values over the selected positions.
+            // positions / num_positions / asso_values are compile-time constants,
+            // so this loop fully unrolls. The idx < len guard mirrors the
+            // generator's char_at()-> 256 -> skip behavior for out-of-range
+            // positions (required: arbitrary positions may exceed a key's length).
+            std::size_t h = len;
+            for (std::uint8_t i = 0; i < phf.num_positions; ++i) {
+                std::uint8_t pos = phf.positions[i];
+                std::size_t idx = (pos == key_selector_detail::POS_LAST_CHAR)
+                                  ? (len - std::size_t{1})
+                                  : static_cast<std::size_t>(pos);
+                if (idx < len) {
+                    h += phf.asso_values[i][static_cast<unsigned char>(p[idx])];
+                }
+            }
+            slot = h & (table_size - 1);
+        }
+
+        std::uint8_t ki = phf.slot_to_key[slot];
+        if (ki >= N) { return N; }
+        if (phf.slot_key_len[slot] != len) { return N; }
+        if (!key_selector_detail::compare_key_bytes<max_key_len>(
+                p, phf.slot_key_bytes[slot].data(), len)) { return N; }
+        return ki;
+    }
+
+    /**
+     * Look up a JSON key. rjs must point just after an opening quote in a padded
+     * simdjson buffer. Returns the selector index in [0, N) on match, or N on miss.
+     * The key length is recovered with a SIMD scan for the closing quote; callers
+     * that already know the length should use the (p, len) overload above.
+     */
+    static simdjson_really_inline std::size_t match_raw(raw_json_string rjs) noexcept {
+        const char* p = rjs.raw();
+        if constexpr (window.ok) {
+            // One 8-bit window picks the candidate; verifying the candidate's
+            // bytes and its closing '"' confirms the full key, so the length scan
+            // is unnecessary. The window read is in bounds (padding), and the
+            // candidate length is at most max_key_len.
+            std::uint8_t ki = window.window_to_key[
+                key_selector_detail::read_window(p, window.byte_offset, window.shift)];
+            if (ki >= N) { return N; }
+            return key_selector_detail::match_window_candidate(
+                p, ki, window, std::make_index_sequence<N>{});
+        }
+        return match_raw(p, key_selector_detail::scan_key_length<max_key_len>(p));
+    }
+
+    /** Return the key text at selector index i (i in [0, N)). */
+    static constexpr std::string_view key_at(std::size_t i) noexcept {
+        return keys[i];
+    }
+
+    /**
+     * Return a complete, human-readable, multi-line description of how this
+     * selector classifies a key: which algorithm was selected at compile time
+     * (single 8-bit window, gperf-style perfect hash, or hash-and-displace), the
+     * exact bytes/positions it inspects, and the contents of the lookup tables
+     * (which window bytes or hash slots map to which key). The text mirrors what
+     * match_raw() does step by step.
+     *
+     * Everything it reports is derived from the compile-time tables, so describe()
+     * is itself usable in a constant expression when the standard library supports
+     * constexpr std::string (__cpp_lib_constexpr_string):
+     *
+     *   static_assert(!key_selector<"name", "city">::describe().empty());
+     *
+     * It allocates a std::string and is meant for documentation, debugging and
+     * tests, not for any hot path.
+     */
+    static SIMDJSON_CONSTEXPR_STRING std::string describe() {
+        std::string s;
+        s += "key_selector: ";
+        key_selector_detail::append_uint(s, N);
+        s += " keys, max key length ";
+        key_selector_detail::append_uint(s, max_key_len);
+        s += "\nkeys:\n";
+        for (std::size_t i = 0; i < N; ++i) {
+            s += "  [";
+            key_selector_detail::append_uint(s, i);
+            s += "] \"";
+            s += keys[i];
+            s += "\" (length ";
+            key_selector_detail::append_uint(s, keys[i].size());
+            s += ")\n";
+        }
+        if constexpr (window.ok) {
+            // Mirrors the window fast path of match_raw().
+            s += "algorithm: single 8-bit window\n";
+            s += "  step 1: read 2 bytes at offset ";
+            key_selector_detail::append_uint(s, window.byte_offset);
+            s += ", interpret them as a little-endian 16-bit value, shift right by ";
+            key_selector_detail::append_uint(s, window.shift);
+            s += " bits, and keep the low 8 bits\n";
+            s += "  step 2: map that byte through a 256-entry table to a key index (";
+            key_selector_detail::append_uint(s, N);
+            s += " means no match):\n";
+            for (std::size_t b = 0; b < 256; ++b) {
+                if (window.window_to_key[b] < N) {
+                    s += "    byte ";
+                    key_selector_detail::append_byte(s, static_cast<unsigned>(b));
+                    s += " -> key ";
+                    key_selector_detail::append_uint(s, window.window_to_key[b]);
+                    s += "\n";
+                }
+            }
+            s += "  step 3: confirm the candidate by checking the closing quote sits at the key's length and comparing the key bytes\n";
+        } else {
+            // Mirrors the perfect-hash path of match_raw().
+            if constexpr (phf.num_positions == key_selector_detail::HD_MODE) {
+                s += "algorithm: hash-and-displace perfect hash\n";
+                s += "  step 1: bucket = (first_byte + last_byte*3 + length*17) mod 256\n";
+                s += "  step 2: keyhash = base-31 rolling hash of the length and the first ";
+                key_selector_detail::append_uint(s, phf.hd_hash_variant);
+                s += " bytes\n";
+                s += "  step 3: slot = (displacement[bucket] + keyhash) mod ";
+                key_selector_detail::append_uint(s, table_size);
+                s += "\n  step 4: slot_to_key[slot] gives the candidate key index\n";
+                s += "  per-key derivation:\n";
+                for (std::size_t i = 0; i < N; ++i) {
+                    std::string_view k = keys[i];
+                    std::size_t bucket = key_selector_detail::hd_bucket_hash(k);
+                    std::size_t kh = (phf.hd_hash_variant == key_selector_detail::HD_HASH_2BYTE_FLAG)
+                        ? key_selector_detail::hd_key_hash_2(k)
+                        : key_selector_detail::hd_key_hash_4(k);
+                    std::size_t slot = (phf.asso_values[0][bucket] + kh) & (table_size - 1);
+                    s += "    \"";
+                    s += k;
+                    s += "\": bucket=";
+                    key_selector_detail::append_uint(s, bucket);
+                    s += " displacement=";
+                    key_selector_detail::append_uint(s, phf.asso_values[0][bucket]);
+                    s += " keyhash=";
+                    key_selector_detail::append_uint(s, kh);
+                    s += " slot=";
+                    key_selector_detail::append_uint(s, slot);
+                    s += "\n";
+                }
+            } else {
+                s += "algorithm: gperf-style perfect hash over ";
+                key_selector_detail::append_uint(s, phf.num_positions);
+                s += " character position(s)\n";
+                s += "  step 1: h = key length\n";
+                s += "  step 2: for each position below, add its association value for the key byte there (a position past the key's length contributes 0):\n";
+                for (std::size_t i = 0; i < phf.num_positions; ++i) {
+                    s += "    position ";
+                    if (phf.positions[i] == key_selector_detail::POS_LAST_CHAR) {
+                        s += "last character";
+                    } else {
+                        s += "byte index ";
+                        key_selector_detail::append_uint(s, phf.positions[i]);
+                    }
+                    s += "\n";
+                }
+                s += "  step 3: slot = h mod ";
+                key_selector_detail::append_uint(s, table_size);
+                s += " (a power of two, applied as a bitmask)\n";
+                s += "  step 4: slot_to_key[slot] gives the candidate key index\n";
+                s += "  per-key derivation:\n";
+                for (std::size_t i = 0; i < N; ++i) {
+                    std::string_view k = keys[i];
+                    std::size_t h = k.size();
+                    for (std::size_t pi = 0; pi < phf.num_positions; ++pi) {
+                        std::size_t pos = phf.positions[pi];
+                        std::size_t idx = (pos == key_selector_detail::POS_LAST_CHAR)
+                                          ? (k.size() - 1) : pos;
+                        if (idx < k.size()) {
+                            h += phf.asso_values[pi][static_cast<unsigned char>(k[idx])];
+                        }
+                    }
+                    std::size_t slot = h & (table_size - 1);
+                    s += "    \"";
+                    s += k;
+                    s += "\": h=";
+                    key_selector_detail::append_uint(s, h);
+                    s += " slot=";
+                    key_selector_detail::append_uint(s, slot);
+                    s += "\n";
+                }
+            }
+            s += "  occupied slots (slot -> key):\n";
+            for (std::size_t slot = 0; slot < table_size; ++slot) {
+                if (phf.slot_to_key[slot] < N) {
+                    s += "    slot ";
+                    key_selector_detail::append_uint(s, slot);
+                    s += " -> key ";
+                    key_selector_detail::append_uint(s, phf.slot_to_key[slot]);
+                    s += " (\"";
+                    s += keys[phf.slot_to_key[slot]];
+                    s += "\", length ";
+                    key_selector_detail::append_uint(s, phf.slot_key_len[slot]);
+                    s += ")\n";
+                }
+            }
+            s += "  confirm the candidate by checking the key length matches and comparing the key bytes\n";
+        }
+        return s;
+    }
+};
+
+namespace key_selector_detail {
+template <typename> struct is_key_selector : std::false_type {};
+template <constevalutil::fixed_string... Keys>
+struct is_key_selector<key_selector<Keys...>> : std::true_type {};
+} // namespace key_selector_detail
+
+/**
+ * Matches any instantiation of key_selector<Keys...>. Used to constrain
+ * object::for_each so that passing a non-selector type yields a clear
+ * constraint error rather than a cascade of failures inside for_each.
+ */
+template <typename T>
+concept key_selector_type = key_selector_detail::is_key_selector<T>::value;
+
+} // namespace ondemand
+} // namespace ppc64
+} // namespace simdjson
+
+#endif // SIMDJSON_SUPPORTS_CONCEPTS
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+/* end file simdjson/generic/ondemand/key_selector.h for ppc64 */
 /* including simdjson/generic/ondemand/object.h for ppc64: #include "simdjson/generic/ondemand/object.h" */
 /* begin file simdjson/generic/ondemand/object.h for ppc64 */
 #ifndef SIMDJSON_GENERIC_ONDEMAND_OBJECT_H
@@ -122526,6 +155690,7 @@ public:
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/implementation_simdjson_result_base.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/key_selector.h" */
 /* amalgamation skipped (editor-only): #include <vector> */
 /* amalgamation skipped (editor-only): #if SIMDJSON_STATIC_REFLECTION && SIMDJSON_SUPPORTS_CONCEPTS */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_string_builder.h"  // for constevalutil::fixed_string */
@@ -122536,6 +155701,114 @@ namespace simdjson {
 namespace ppc64 {
 namespace ondemand {

+#if SIMDJSON_SUPPORTS_CONCEPTS
+/**
+ * Result of object::for_each: the first error encountered (SUCCESS if none) and
+ * the number of distinct selector keys that matched during the walk. A
+ * matched_count equal to Selector::size() means every selected key was present
+ * in the object. Implicitly converts to error_code so existing callers that only
+ * care about the error (including SIMDJSON_TRY and the test ASSERT_* macros) keep
+ * working unchanged.
+ *
+ * On error paths (a handler returns a non-SUCCESS error_code, or a direct-target
+ * deserialization via value::get fails), matched_count reflects only the number
+ * of distinct selected keys that were successfully handled *before* the failure;
+ * the failing key is not counted. This is the current behavior.
+ *
+ * Marked [[nodiscard]]: for_each surfaces parse/type errors (from a matched
+ * value, a returning handler, or the structural walk itself) only through this
+ * result, so it must not be silently dropped. This matters most in builds
+ * without exceptions, where it is the only channel for those errors. To
+ * deliberately ignore it, assign to a variable or cast to void.
+ */
+struct [[nodiscard]] for_each_result {
+  error_code error{SUCCESS};
+  std::size_t matched_count{0};
+  constexpr operator error_code() const noexcept { return error; }
+};
+
+namespace key_selector_for_each_detail {
+/**
+ * A per-key handler for the variadic object::for_each is either:
+ *   - an invocable taking a value (run custom logic for that field), or
+ *   - a deserialization target T, in which case the matched value is assigned
+ *     directly via value::get(T&).
+ * The target form lets callers bind fields straight to variables without a
+ * lambda per key -- e.g. obj.for_each<"name","city","age">(name, city, age) --
+ * while still allowing a lambda in any position when custom logic is needed
+ * (for example to descend into a nested object).
+ */
+template <typename H>
+concept field_handler =
+    std::is_invocable_v<std::remove_reference_t<H>&, value> ||
+    ::simdjson::deserializable<std::remove_cvref_t<H>, value>;
+
+/**
+ * noexcept-ness of handling a single handler: nothrow-invocability for the
+ * callback form, or nothrow-deserializability for the direct-target form
+ * (builtin scalar/string targets are noexcept; custom targets follow their
+ * own tag_invoke noexcept specification).
+ */
+template <typename H>
+inline constexpr bool nothrow_field_handler_v =
+    std::is_invocable_v<std::remove_reference_t<H>&, value>
+        ? std::is_nothrow_invocable_v<std::remove_reference_t<H>&, value>
+        : ::simdjson::nothrow_deserializable<std::remove_cvref_t<H>, value>;
+
+/**
+ * nothrow_field_handler_v folded over a whole handler pack. The variadic
+ * for_each overloads spell their conditional-noexcept specifier with this
+ * single id-expression rather than an inline fold expression. MSVC (VS18)
+ * mishandles a fold-expression that appears directly in the noexcept-specifier
+ * of an out-of-line template definition: it fails to recognize the definition's
+ * specifier as matching the in-class declaration's and reports a spurious
+ * C2382 ("redefinition; different exception specifications"). Naming the fold
+ * here keeps the specifier a plain identifier, which matches reliably.
+ */
+template <typename... Handlers>
+inline constexpr bool nothrow_field_handlers_v =
+    (nothrow_field_handler_v<Handlers> && ...);
+} // namespace key_selector_for_each_detail
+#endif
+
+/**
+ * An opaque snapshot of an object's scanning position, obtained from
+ * object::get_current_position() and consumed by object::revert_position().
+ *
+ * This bundles the raw token position with the iteration depth that was
+ * live at the moment of capture: a scalar field value (e.g. a number)
+ * unwinds back to the object's own depth once fully consumed, while a
+ * compound value (array/object) not yet fully consumed stays one level
+ * deeper. The correct depth to restore to therefore depends on what was
+ * live when the snapshot was taken, not on the object itself, which is
+ * why both pieces travel together here rather than being recomputed later.
+ *
+ * The members are private and only constructible by object itself: a
+ * hand-built value here would let revert_position() jump to an arbitrary,
+ * unvalidated position and depth, and this type carries no way to check
+ * that a given snapshot still refers to a live object. Only ever pass
+ * along a value you got from get_current_position(), and only back to the
+ * same object, before that object has been reset() or the parser has
+ * iterate()d a new document: neither invalidates a snapshot in a way this
+ * type can detect.
+ */
+struct object_position {
+  /**
+   * Default-constructed so a variable can be declared and assigned later,
+   * matching e.g. document()/object(). Not a valid position to revert to.
+   */
+  simdjson_inline object_position() noexcept = default;
+
+private:
+  token_position position{};
+  depth_t depth{};
+
+  simdjson_inline object_position(token_position position_, depth_t depth_) noexcept
+    : position(position_), depth(depth_) {}
+
+  friend class object;
+};
+
 /**
  * A forward-only JSON object field iterator.
  */
@@ -122554,8 +155827,19 @@ public:
    * Using the iterator directly is also possible but error-prone and discouraged. In particular,
    * you must dereference the iterator exactly once per iteration (before calling '++').
    * Doing otherwise is unsafe and may lead to errors. You are responsible for ensuring
+   *
+   * With SIMDJSON_DEVELOPMENT_CHECKS, the iterator locks this object while it is
+   * alive, so that reentrant access (find_field(), reset(), ...) is reported as
+   * OUT_OF_ORDER_ITERATION.
+   */
+  simdjson_inline simdjson_result<object_iterator> begin() & noexcept;
+  /**
+   * Get an iterator to the start of a temporary object, e.g., `v.get_object().begin()`.
+   *
+   * The iterator does not depend on the object instance and may outlive it, so
+   * it does not lock it.
    */
-  simdjson_inline simdjson_result<object_iterator> begin() noexcept;
+  simdjson_inline simdjson_result<object_iterator> begin() && noexcept;
   simdjson_inline simdjson_result<object_iterator> end() noexcept;
   /**
    * Look up a field by name on an object (order-sensitive). By order-sensitive, we mean that
@@ -122567,10 +155851,11 @@ public:
    *
    * ```cpp
    * simdjson::ondemand::parser parser;
-   * auto obj = parser.parse(R"( { "x": 1, "y": 2, "z": 3 } )"_padded);
-   * double z = obj.find_field("z");
-   * double y = obj.find_field("y");
-   * double x = obj.find_field("x");
+   * auto json = R"( { "x": 1, "y": 2, "z": 3 } )"_padded;
+   * auto doc = parser.iterate(json);
+   * double z = doc.find_field("z");
+   * double y = doc.find_field("y");
+   * double x = doc.find_field("x");
    * ```
    * If you have multiple fields with a matching key ({"x": 1,  "x": 1}) be mindful
    * that only one field is returned.
@@ -122643,6 +155928,100 @@ public:
   /** @overload simdjson_inline simdjson_result<value> find_field_unordered(std::string_view key) & noexcept; */
   simdjson_inline simdjson_result<value> operator[](std::string_view key) && noexcept;

+#if SIMDJSON_SUPPORTS_CONCEPTS
+  /**
+   * Walk this object once and invoke on_match(selector_index, value) for each
+   * field whose key is in the compile-time key_selector Selector, in JSON order
+   * (first occurrence of a duplicate key wins). Iteration stops once all
+   * Selector::size() keys have matched or the object ends. The value is consumed
+   * in place, so this is a low-overhead way to extract a known set of fields
+   * regardless of their order in the JSON.
+   *
+   * Like other object iteration in simdjson, for_each consumes the object by
+   * advancing the underlying iterator state; after the call the same object
+   * instance should not be used for further field access or iteration.
+   *
+   * Usage:
+   *   using sel_t = ondemand::key_selector<"id", "text", "user">;
+   *   obj.for_each<sel_t>([&](std::size_t i, ondemand::value v) {
+   *     switch (i) { case 0: ...; case 1: ...; }
+   *   });
+   *
+   * Limitations (see key_selector): each key must be at most 63 characters long,
+   * and the number of keys should be moderate (hard limit 255; a handful is
+   * best, as the compile-time perfect hash may fail or slow compilation for
+   * large key sets). The keys must be distinct, non-empty, and free of backslash, double-quote and
+   * null bytes.
+   *
+   * The callback may return either void or an error_code. When it returns an
+   * error_code, the walk stops at the first non-SUCCESS result and that error is
+   * returned, which lets the callback surface value-parse errors.
+   *
+   * This function is conditionally noexcept: it is noexcept exactly when invoking
+   * the callback is noexcept. The callback runs inside this frame, so a throwing
+   * callback (e.g. one using the exception-throwing conversions like
+   * std::string_view(value) or uint64_t(value)) makes for_each potentially
+   * throwing too -- the exception propagates to the caller instead of crossing a
+   * noexcept boundary and calling std::terminate.
+   *
+   * @returns a for_each_result holding the first error encountered while walking
+   *          the object (including any error returned by the callback, SUCCESS if
+   *          none) and the number of distinct selector keys that matched. The
+   *          result converts implicitly to error_code, so callers that only need
+   *          the error can ignore the count.
+   */
+  template <typename Selector, typename Func>
+    requires key_selector_type<Selector> &&
+             std::is_invocable_v<Func&, std::size_t, value>
+  simdjson_inline for_each_result for_each(Func&& on_match)
+      noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, value>);
+
+  /**
+   * Variadic per-key form. Provide exactly one handler per key in the Selector
+   * (compiler-enforced). Handlers are processed in JSON document order for the
+   * matching keys. Each handler is either:
+   *   - a deserialization target (a variable), in which case the matched value
+   *     is assigned to it via value::get -- no lambda required; or
+   *   - an invocable taking the ondemand::value (for custom logic such as
+   *     descending into a nested object). It may return void or error_code;
+   *     returning error_code lets you surface parse/type errors.
+   * The two styles may be mixed freely, one handler per key.
+   *
+   * Example (bind fields straight to variables):
+   *   using fields = ondemand::key_selector<"name", "city", "age">;
+   *   obj.for_each<fields>(name, city, age);
+   *
+   * Example (mixing a target and a lambda):
+   *   obj.for_each<ondemand::key_selector<"id", "user">>(
+   *     id,                                          // assigned via value::get
+   *     [&](ondemand::value v){ u = read_user(v); }  // custom logic
+   *   );
+   *
+   * The index-based single-callback form (taking (size_t, value)) remains
+   * available for shared-state or more complex per-key logic.
+   */
+  template <typename Selector, typename... Handlers>
+    requires key_selector_type<Selector> &&
+             (sizeof...(Handlers) == Selector::size()) &&
+             (key_selector_for_each_detail::field_handler<Handlers> && ...)
+  simdjson_inline for_each_result for_each(Handlers&&... on_match)
+      noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
+  /**
+   * Direct-key shorthand. Equivalent to for_each<key_selector<Keys...>>(...).
+   * Lets you write the keys inline without a separate using/alias, binding each
+   * field straight to a variable (or a lambda, see the Selector form above):
+   *
+   *   obj.for_each<"name", "city", "age">(name, city, age);
+   */
+  template <constevalutil::fixed_string... Keys, typename... Handlers>
+    requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+             (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+             (key_selector_for_each_detail::field_handler<Handlers> && ...)
+  simdjson_inline for_each_result for_each(Handlers&&... on_match)
+      noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+#endif
+
   /**
    * Get the value associated with the given JSON pointer. We use the RFC 6901
    * https://tools.ietf.org/html/rfc6901 standard, interpreting the current node
@@ -122719,6 +156098,34 @@ public:
    * @returns true if the object contains some elements (not empty)
    */
   inline simdjson_result<bool> reset() & noexcept;
+  /**
+   * Get an opaque token representing the object's current scanning position.
+   * Pass it to revert_position() to return to this exact point later, without
+   * paying the cost of a full reset() and re-scan from the beginning.
+   *
+   * A typical use is an optional field that may or may not be next: capture
+   * the position, attempt find_field(), and on NO_SUCH_FIELD, revert_position()
+   * instead of reset() so that fields already consumed are not rescanned.
+   *
+   * The returned token is only valid for this object, and only until it is
+   * reset() or the parser iterate()s a new document; using it after either
+   * is undefined behavior (see object_position).
+   *
+   * @returns An opaque position token.
+   */
+  simdjson_inline object_position get_current_position() const noexcept;
+  /**
+   * Return the object's scanning position to a snapshot previously obtained
+   * from get_current_position(). Unlike reset(), this does not rescan the
+   * object from the beginning: fields before the captured position remain
+   * consumed, and scanning resumes exactly where the snapshot was captured.
+   *
+   * @param position A snapshot previously returned by get_current_position(),
+   *        for this same object.
+   * @returns SUCCESS, or OUT_OF_ORDER_ITERATION if called during active
+   *          iteration (SIMDJSON_DEVELOPMENT_CHECKS builds only).
+   */
+  simdjson_inline error_code revert_position(object_position position) noexcept;
   /**
    * This method scans the beginning of the object and checks whether the
    * object is empty.
@@ -122764,7 +156171,7 @@ public:
    */
   template <typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out)
-     noexcept(custom_deserializable<T, object> ? nothrow_custom_deserializable<T, object> : true) {
+     noexcept(nothrow_gettable<T, object>) {
     static_assert(custom_deserializable<T, object>);
     return deserialize(*this, out);
   }
@@ -122776,7 +156183,7 @@ public:
    */
   template <typename T>
   simdjson_inline simdjson_result<T> get()
-    noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+    noexcept(nothrow_gettable<T, object>)
   {
     static_assert(std::is_default_constructible<T>::value, "The specified type is not default constructible.");
     T out{};
@@ -122828,10 +156235,18 @@ protected:
   simdjson_warn_unused simdjson_inline error_code find_field_raw(const std::string_view key) noexcept;

   value_iterator iter{};
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  bool locked{false};
+  simdjson_inline void set_locked(bool _locked) noexcept;
+#endif

   friend class value;
   friend class document;
   friend struct simdjson_result<object>;
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  friend class object_iterator;
+  friend struct simdjson_result<object_iterator>;
+#endif
 };

 } // namespace ondemand
@@ -122847,7 +156262,8 @@ public:
   simdjson_inline simdjson_result(error_code error) noexcept; ///< @private
   simdjson_inline simdjson_result() noexcept = default;

-  simdjson_inline simdjson_result<ppc64::ondemand::object_iterator> begin() noexcept;
+  simdjson_inline simdjson_result<ppc64::ondemand::object_iterator> begin() & noexcept;
+  simdjson_inline simdjson_result<ppc64::ondemand::object_iterator> begin() && noexcept;
   simdjson_inline simdjson_result<ppc64::ondemand::object_iterator> end() noexcept;
   simdjson_inline simdjson_result<ppc64::ondemand::value> find_field(std::string_view key) & noexcept;
   simdjson_inline simdjson_result<ppc64::ondemand::value> find_field(std::string_view key) && noexcept;
@@ -122865,6 +156281,8 @@ public:
 #endif
   simdjson_inline error_code for_each_at_path_with_wildcard(std::string_view json_path, Func&& callback) noexcept;
   inline simdjson_result<bool> reset() noexcept;
+  inline simdjson_result<ppc64::ondemand::object_position> get_current_position() noexcept;
+  inline error_code revert_position(ppc64::ondemand::object_position position) noexcept;
   inline simdjson_result<bool> is_empty() noexcept;
   inline simdjson_result<size_t> count_fields() & noexcept;
   inline simdjson_result<std::string_view> raw_json() noexcept;
@@ -122872,7 +156290,7 @@ public:
   // TODO: move this code into object-inl.h

   template<typename T>
-  simdjson_inline simdjson_result<T> get() noexcept {
+  simdjson_inline simdjson_result<T> get() noexcept(nothrow_gettable<T, ppc64::ondemand::object>) {
     if (error()) { return error(); }
     if constexpr (std::is_same_v<T, ppc64::ondemand::object>) {
       return first;
@@ -122880,7 +156298,7 @@ public:
     return first.get<T>();
   }
   template<typename T>
-  simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept {
+  simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept(nothrow_gettable<T, ppc64::ondemand::object>) {
     if (error()) { return error(); }
     if constexpr (std::is_same_v<T, ppc64::ondemand::object>) {
       out = first;
@@ -122890,6 +156308,39 @@ public:
     return SUCCESS;
   }

+  /**
+   * Forwards to object::for_each on the underlying object, so error-code-style
+   * chains (e.g. doc["x"].get_object()) can call for_each without first
+   * extracting the object. If this result holds an error, that error is returned
+   * (with a zero match count) and the callback is not invoked. See
+   * object::for_each for the semantics.
+   */
+  template <typename Selector, typename Func>
+    requires ppc64::ondemand::key_selector_type<Selector> &&
+             std::is_invocable_v<Func&, std::size_t, ppc64::ondemand::value>
+  simdjson_inline ppc64::ondemand::for_each_result for_each(Func&& on_match)
+      noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, ppc64::ondemand::value>);
+
+  /**
+   * Forwarding overload for the variadic per-key form (targets and/or lambdas).
+   */
+  template <typename Selector, typename... Handlers>
+    requires ppc64::ondemand::key_selector_type<Selector> &&
+             (sizeof...(Handlers) == Selector::size()) &&
+             (ppc64::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+  simdjson_inline ppc64::ondemand::for_each_result for_each(Handlers&&... on_match)
+      noexcept(ppc64::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
+  /**
+   * Forwarding overload for the direct-key variadic form.
+   */
+  template <constevalutil::fixed_string... Keys, typename... Handlers>
+    requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+             (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+             (ppc64::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+  simdjson_inline ppc64::ondemand::for_each_result for_each(Handlers&&... on_match)
+      noexcept(ppc64::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
 #if SIMDJSON_STATIC_REFLECTION
   // TODO: move this code into object-inl.h
   template<constevalutil::fixed_string... FieldNames, typename T>
@@ -122930,6 +156381,15 @@ public:
    */
   simdjson_inline object_iterator() noexcept = default;

+#if SIMDJSON_DEVELOPMENT_CHECKS
+   simdjson_inline ~object_iterator() noexcept;
+
+   simdjson_inline object_iterator(object_iterator&&) noexcept;
+   simdjson_inline object_iterator& operator=(object_iterator&&) noexcept;
+   simdjson_inline object_iterator(const object_iterator&) noexcept;
+   simdjson_inline object_iterator& operator=(const object_iterator&) noexcept;
+#endif
+
   //
   // Iterator interface
   //
@@ -122949,6 +156409,9 @@ public:
 private:
 #if SIMDJSON_DEVELOPMENT_CHECKS
    bool has_been_referenced{false};
+   object* parent{nullptr};
+
+   simdjson_inline object_iterator(const value_iterator &_iter, object* _parent) noexcept;
 #endif
   /**
    * The underlying JSON iterator.
@@ -122994,6 +156457,191 @@ public:

 #endif // SIMDJSON_GENERIC_ONDEMAND_OBJECT_ITERATOR_H
 /* end file simdjson/generic/ondemand/object_iterator.h for ppc64 */
+/* including simdjson/generic/ondemand/ranges.h for ppc64: #include "simdjson/generic/ondemand/ranges.h" */
+/* begin file simdjson/generic/ondemand/ranges.h for ppc64 */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_RANGES_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_ONDEMAND_RANGES_H */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/field.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#if SIMDJSON_SUPPORTS_RANGES
+
+namespace simdjson {
+namespace ppc64 {
+namespace ondemand {
+
+/**
+ * A ranges-compatible iterator adapter for JSON arrays.
+ *
+ * Wraps array_iterator to satisfy std::input_iterator by providing:
+ * - const operator* (via mutable internal state)
+ * - post-increment operator
+ * - iterator_concept tag
+ *
+ * The mutable approach is standard for single-pass input iterators that
+ * read from external sources (similar to std::istream_iterator).
+ */
+class array_range_iterator {
+public:
+  using iterator_concept = std::input_iterator_tag;
+  using value_type = simdjson_result<value>;
+  using reference = simdjson_result<value>;
+  using difference_type = std::ptrdiff_t;
+
+  simdjson_inline array_range_iterator() noexcept = default;
+  simdjson_inline explicit array_range_iterator(array_iterator iter) noexcept;
+
+  /**
+   * Get the current element. Const-qualified for std::indirectly_readable;
+   * internally delegates to the mutable wrapped iterator.
+   */
+  simdjson_inline simdjson_result<value> operator*() const noexcept;
+  simdjson_inline array_range_iterator& operator++() noexcept;
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+  simdjson_inline void operator++(int) noexcept;
+SIMDJSON_POP_DISABLE_WARNINGS
+  /**
+   * Comparison delegates to array_iterator::operator==, which checks
+   * whether the underlying parser has finished the array (depth-based).
+   */
+  simdjson_inline friend bool operator==(const array_range_iterator& a,
+                                         const array_range_iterator& b) noexcept {
+    return a.iter_ == b.iter_;
+  }
+
+private:
+  mutable array_iterator iter_{};
+};
+
+/**
+ * A std::ranges::view over a JSON array.
+ *
+ * Wraps an ondemand::array and exposes begin()/end() that return
+ * array_range_iterator (satisfying std::input_iterator), enabling
+ * use with std::views::transform and other range adaptors.
+ *
+ * If the array's begin() returns an error (only possible under
+ * SIMDJSON_DEVELOPMENT_CHECKS), the range will be empty and error()
+ * will return the error code.
+ *
+ * Usage:
+ *   ondemand::parser parser;
+ *   auto doc = parser.iterate(json);
+ *   auto arr = doc.get_array().value();
+ *   for (auto elem : ondemand::get_range(arr)) { ... }
+ */
+class array_range {
+public:
+  simdjson_inline array_range() noexcept = default;
+  simdjson_inline explicit array_range(array& arr) noexcept;
+
+  simdjson_inline array_range_iterator begin() noexcept;
+  simdjson_inline array_range_iterator end() noexcept;
+
+  /** Returns SUCCESS if the range was created successfully, or the error code otherwise. */
+  simdjson_inline error_code error() const noexcept { return error_; }
+
+private:
+  array_iterator begin_{};
+  array_iterator end_{};
+  error_code error_{SUCCESS};
+};
+
+/**
+ * A ranges-compatible iterator adapter for JSON objects.
+ *
+ * Wraps object_iterator to satisfy std::input_iterator, yielding
+ * simdjson_result<field> elements (key-value pairs).
+ */
+class object_range_iterator {
+public:
+  using iterator_concept = std::input_iterator_tag;
+  using value_type = simdjson_result<field>;
+  using reference = simdjson_result<field>;
+  using difference_type = std::ptrdiff_t;
+
+  simdjson_inline object_range_iterator() noexcept = default;
+  simdjson_inline explicit object_range_iterator(object_iterator iter) noexcept;
+
+  simdjson_inline simdjson_result<field> operator*() const noexcept;
+  simdjson_inline object_range_iterator& operator++() noexcept;
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+  simdjson_inline void operator++(int) noexcept;
+SIMDJSON_POP_DISABLE_WARNINGS
+  simdjson_inline friend bool operator==(const object_range_iterator& a,
+                                         const object_range_iterator& b) noexcept {
+    return a.iter_ == b.iter_;
+  }
+
+private:
+  mutable object_iterator iter_{};
+};
+
+/**
+ * A std::ranges::view over a JSON object.
+ *
+ * Wraps an ondemand::object and exposes begin()/end() that return
+ * object_range_iterator, enabling use with range adaptors.
+ *
+ * If the object's begin() returns an error, the range will be empty
+ * and error() will return the error code.
+ */
+class object_range {
+public:
+  simdjson_inline object_range() noexcept = default;
+  simdjson_inline explicit object_range(object& obj) noexcept;
+
+  simdjson_inline object_range_iterator begin() noexcept;
+  simdjson_inline object_range_iterator end() noexcept;
+
+  /** Returns SUCCESS if the range was created successfully, or the error code otherwise. */
+  simdjson_inline error_code error() const noexcept { return error_; }
+
+private:
+  object_iterator begin_{};
+  object_iterator end_{};
+  error_code error_{SUCCESS};
+};
+
+/** Get a std::ranges compatible view over a JSON array. */
+simdjson_inline array_range get_range(array& arr) noexcept;
+
+/** Get a std::ranges compatible view over a JSON object (key-value pairs). */
+simdjson_inline object_range get_key_value_range(object& obj) noexcept;
+
+#if SIMDJSON_EXCEPTIONS
+/** Get a std::ranges compatible view, unwrapping the simdjson_result (throws on error). */
+simdjson_inline array_range get_range(simdjson_result<array> result);
+
+/** Get a std::ranges compatible view, unwrapping the simdjson_result (throws on error). */
+simdjson_inline object_range get_key_value_range(simdjson_result<object> result);
+#endif // SIMDJSON_EXCEPTIONS
+
+} // namespace ondemand
+} // namespace ppc64
+} // namespace simdjson
+
+namespace std {
+namespace ranges {
+template<>
+inline constexpr bool enable_view<simdjson::ppc64::ondemand::array_range> = true;
+template<>
+inline constexpr bool enable_view<simdjson::ppc64::ondemand::object_range> = true;
+} // namespace ranges
+} // namespace std
+
+#endif // SIMDJSON_SUPPORTS_RANGES
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_RANGES_H
+/* end file simdjson/generic/ondemand/ranges.h for ppc64 */
 /* including simdjson/generic/ondemand/serialization.h for ppc64: #include "simdjson/generic/ondemand/serialization.h" */
 /* begin file simdjson/generic/ondemand/serialization.h for ppc64 */
 #ifndef SIMDJSON_GENERIC_ONDEMAND_SERIALIZATION_H
@@ -123126,12 +156774,14 @@ inline std::ostream& operator<<(std::ostream& out, simdjson::simdjson_result<sim
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/annotations.h" */
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */

 #include <concepts>
 #include <limits>
 #if SIMDJSON_STATIC_REFLECTION
 #include <meta>
+#include <vector>
 // #include <static_reflection> // for std::define_static_string - header not available yet
 #endif

@@ -123156,10 +156806,25 @@ error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept {

 template <std::floating_point T>
 error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept {
-  double x;
-  SIMDJSON_TRY(val.get_double().get(x));
-  out = static_cast<T>(x);
-  return SUCCESS;
+  if constexpr (std::is_same_v<T, float>) {
+    // Going through binary64 and then rounding to binary32 would round twice
+    // and could produce a value that is not the float nearest to the JSON
+    // number, so we parse to binary32 directly.
+    return val.get_float().get(out);
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  } else if constexpr (std::is_same_v<T, std::float32_t>) {
+    // Same reason as float.
+    float x;
+    SIMDJSON_TRY(val.get_float().get(x));
+    out = static_cast<T>(x);
+    return SUCCESS;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+  } else {
+    double x;
+    SIMDJSON_TRY(val.get_double().get(x));
+    out = static_cast<T>(x);
+    return SUCCESS;
+  }
 }

 template <std::signed_integral T>
@@ -123195,11 +156860,59 @@ template <concepts::constructible_from_string_view T, typename ValT>
 error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(std::is_nothrow_constructible_v<T, std::string_view>) {
   std::string_view str;
   SIMDJSON_TRY(val.get_string().get(str));
-  out = T{str};
+  if constexpr (requires { out.assign(str.data(), str.size()); }) {
+    // Copy straight into out (e.g., std::string): building a temporary and
+    // move-assigning it is markedly slower.
+    out.assign(str.data(), str.size());
+  } else {
+    out = T{str};
+  }
+  return SUCCESS;
+}
+
+#if SIMDJSON_SUPPORTS_CHAR8_T
+// any C++20 char8_t string-like type (can be constructed from std::u8string_view),
+// such as std::u8string
+template <concepts::constructible_from_u8string_view T, typename ValT>
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(std::is_nothrow_constructible_v<T, std::u8string_view>) {
+  std::u8string_view str;
+  SIMDJSON_TRY(val.get_u8string().get(str));
+  if constexpr (requires { out.assign(str.data(), str.size()); }) {
+    // Copy straight into out (e.g., std::u8string), as for std::string above.
+    out.assign(str.data(), str.size());
+  } else {
+    out = T{str};
+  }
   return SUCCESS;
 }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T


+namespace details {
+// Whether to deserialize the elements of the container T directly into a new
+// element (emplace_one(out) and then get) rather than into a temporary that is
+// then moved into the container. A temporary of a trivially copyable type lives
+// in registers and is cheap to move, and value-initializing a small element in
+// place costs more than it saves (GCC zeroes it with rep stos). But moving a
+// large element, or a string (whose deserialization then needs a temporary
+// string and a move assignment of its own), is expensive.
+template <typename T>
+concept deserialize_in_place =
+    concepts::returns_reference<T> && requires(T &c) { c.pop_back(); } &&
+    !std::is_trivially_copyable_v<typename T::value_type> &&
+    (sizeof(typename T::value_type) > 32 || concepts::constructible_from_string_view<typename T::value_type>);
+
+// Removes the last element of the container on destruction while armed.
+template <typename T>
+struct pop_back_guard {
+  T &container;
+  bool armed{true};
+  ~pop_back_guard() {
+    if (armed) { container.pop_back(); }
+  }
+};
+} // namespace details
+
 /**
  * STL containers have several constructors including one that takes a single
  * size argument. Thus, some compilers (Visual Studio) will not be able to
@@ -123223,22 +156936,65 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
     SIMDJSON_TRY(val.get_array().get(arr));
   }

-  for (auto v : arr) {
-    if constexpr (concepts::returns_reference<T>) {
-      if (auto const err = v.get<value_type>().get(concepts::emplace_one(out));
-          err) {
-        // If an error occurs, the empty element that we just inserted gets
-        // removed. We're not using a temp variable because if T is a heavy
-        // type, we want the valid path to be the fast path and the slow path be
-        // the path that has errors in it.
-        if constexpr (requires { out.pop_back(); }) {
-          static_cast<void>(out.pop_back());
+  if constexpr (std::is_same_v<T, std::vector<value_type>> && !std::is_same_v<value_type, bool>) {
+    // Collect the elements in a per-thread scratch vector that keeps its
+    // capacity from call to call, then move them into out after reserving the
+    // exact size: out is allocated once instead of being regrown. A nested
+    // array of the same type finds the scratch busy and takes the paths below.
+    // Prior related work: jsonifier keeps a thread-local vector and sizes the
+    // caller's vector from that element count (parse_impl.hpp,
+    // https://github.com/nihilai-collective/Jsonifier).
+    struct scratch_space {
+      std::vector<value_type> elements{};
+      bool busy{false};
+    };
+    static thread_local scratch_space scratch;
+    if (!scratch.busy && out.empty()) {
+      struct release_scratch {
+        scratch_space &s;
+        T &out;
+        size_t parsed{0};
+        bool complete{false};
+        // On an error or an exception, out gets the elements parsed so far (as
+        // with the loops below), without allocating. Kept out of the hot path.
+        simdjson_never_inline void keep_parsed() noexcept {
+          s.elements.resize(parsed);
+          out.swap(s.elements);
         }
-        return err;
-      }
-    } else {
+        ~release_scratch() {
+          if (simdjson_unlikely(!complete)) { keep_parsed(); }
+          s.elements.clear();
+          // Do not hold on to the memory of a very large array.
+          if (s.elements.capacity() * sizeof(value_type) > (1 << 20)) { std::vector<value_type>().swap(s.elements); }
+          s.busy = false;
+        }
+      } release{scratch, out};
+      scratch.busy = true;
+      for (auto v : arr) {
+        SIMDJSON_TRY(v.get<value_type>(scratch.elements.emplace_back()));
+        release.parsed++;
+      }
+      out.reserve(release.parsed);
+      release.complete = true;
+      for (auto &e : scratch.elements) { out.emplace_back(std::move(e)); }
+      return SUCCESS;
+    }
+  }
+  if constexpr (details::deserialize_in_place<T>) {
+    for (auto v : arr) {
+      auto &slot = concepts::emplace_one(out);
+      // An error or an exception (a user tag_invoke may throw) must not leave
+      // a partially deserialized element behind.
+      details::pop_back_guard<T> guard{out};
+      SIMDJSON_TRY(v.get<value_type>(slot));
+      guard.armed = false;
+    }
+  } else {
+    for (auto v : arr) {
+      // Deserialize into a temporary first: an error or an exception (a user
+      // tag_invoke may throw) must not leave a default-constructed element behind.
       value_type temp;
-      if (auto const err = v.get<value_type>().get(temp); err) {
+      if (auto const err = v.get<value_type>(temp); err) {
         return err;
       }
       concepts::emplace_one(out, std::move(temp));
@@ -123279,7 +157035,7 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
 }

 template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, ppc64::ondemand::object &obj, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, ppc64::ondemand::object &obj, T &out) noexcept(false) {
   using value_type = typename std::remove_cvref_t<T>::mapped_type;

   out.clear();
@@ -123298,21 +157054,21 @@ error_code tag_invoke(deserialize_tag, ppc64::ondemand::object &obj, T &out) noe
 }

 template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, ppc64::ondemand::value &val, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, ppc64::ondemand::value &val, T &out) noexcept(false) {
   ppc64::ondemand::object obj;
   SIMDJSON_TRY(val.get_object().get(obj));
   return simdjson::deserialize(obj, out);
 }

 template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, ppc64::ondemand::document &doc, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, ppc64::ondemand::document &doc, T &out) noexcept(false) {
   ppc64::ondemand::object obj;
   SIMDJSON_TRY(doc.get_object().get(obj));
   return simdjson::deserialize(obj, out);
 }

 template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, ppc64::ondemand::document_reference &doc, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, ppc64::ondemand::document_reference &doc, T &out) noexcept(false) {
   ppc64::ondemand::object obj;
   SIMDJSON_TRY(doc.get_object().get(obj));
   return simdjson::deserialize(obj, out);
@@ -123323,10 +157079,6 @@ error_code tag_invoke(deserialize_tag, ppc64::ondemand::document_reference &doc,
  * This CPO (Customization Point Object) will help deserialize into
  * smart pointers.
  *
- * If constructing T is nothrow, this conversion should be nothrow as well since
- * we return MEMALLOC if we're not able to allocate memory instead of throwing
- * the error message.
- *
  * @tparam T The type inside the smart pointer
  * @tparam ValT document/value type
  * @param val document/value
@@ -123334,7 +157086,7 @@ error_code tag_invoke(deserialize_tag, ppc64::ondemand::document_reference &doc,
  * @return status of the conversion
  */
 template <concepts::smart_pointer T, typename ValT>
-error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(nothrow_deserializable<typename std::remove_cvref_t<T>::element_type, ValT>) {
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
   using element_type = typename std::remove_cvref_t<T>::element_type;

   // For better error messages, don't use these as constraints on
@@ -123346,12 +157098,13 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(nothrow_deser
       std::is_default_constructible_v<element_type>,
       "The specified type inside the unique_ptr must default constructible.");

-  auto ptr = new (std::nothrow) element_type();
-  if (ptr == nullptr) {
+  // Own the allocation before get(): a user tag_invoke may throw.
+  std::unique_ptr<element_type> ptr(new (std::nothrow) element_type());
+  if (!ptr) {
     return MEMALLOC;
   }
   SIMDJSON_TRY(val.template get<element_type>(*ptr));
-  out.reset(ptr);
+  out = std::move(ptr);
   return SUCCESS;
 }

@@ -123383,53 +157136,595 @@ error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept(nothrow_deser

 template <typename T>
 constexpr bool user_defined_type = (std::is_class_v<T>
-&& !std::is_same_v<T, std::string> && !std::is_same_v<T, std::string_view> && !concepts::optional_type<T> &&
-!concepts::appendable_containers<T>);
+&& !std::is_same_v<T, std::string> && !std::is_same_v<T, std::string_view>
+#if SIMDJSON_SUPPORTS_CHAR8_T
+// The char8_t string types are class types with no reflectable members, so
+// without this they would be taken for user structs and the reflection
+// overload below would compete with the u8 string overload in
+// std_deserialize.h, making every tag_invoke call on them ambiguous.
+&& !std::is_same_v<T, std::u8string> && !std::is_same_v<T, std::u8string_view>
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+&& !concepts::optional_type<T> &&
+!concepts::appendable_containers<T>
+// simdjson's own types (array, object, value, raw_json_string, number,
+// document, document_reference) have dedicated get<T>() specializations and
+// must never go through reflection.
+&& !is_builtin_deserializable_v<T>
+&& !std::is_same_v<T, ppc64::ondemand::number>
+&& !std::is_same_v<T, ppc64::ondemand::document>
+&& !std::is_same_v<T, ppc64::ondemand::document_reference>);
+
+
+// key_selector_reflection_detail is defined unconditionally (it only requires
+// static reflection). It provides the compile-time machinery for building a
+// key_selector from a struct's members, the per-member helpers shared by every
+// deserialization path (they implement the annotations of annotations.h), the
+// ordered per-member path used by the opt-out build (see
+// deserialize_struct_ordered below), and the scan used for structs annotated
+// with deny_unknown_fields and as an automatic fallback (see
+// deserialize_struct_scan and keys_fit_selector below).
+namespace key_selector_reflection_detail {
+
+// A member participates if it is public, non-const, and not annotated with skip
+// or skip_deserializing.
+consteval bool is_eligible_member(std::meta::info mem) {
+  return !std::meta::is_const(mem) && std::meta::is_public(mem)
+      && !simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_tag)
+      && !simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_deserializing_tag);
+}
+
+// A field is a data member reached from T through a chain of members: usually a
+// direct member of T (a path of length one), or a member of a member annotated
+// with flatten (the members of a flattened structure are fields of the enclosing
+// one). A field is identified by the type member_path<m1, m2, ..., leaf>.
+template <std::meta::info First, std::meta::info... Rest>
+struct member_path {
+  // The data member holding the value; its annotations drive (de)serialization.
+  static constexpr std::meta::info leaf = [] {
+    std::meta::info members[] = {First, Rest...};
+    return members[sizeof...(Rest)];
+  }();
+  template <typename T>
+  static simdjson_inline constexpr auto &get(T &obj) noexcept {
+    if constexpr (sizeof...(Rest) == 0) {
+      return obj.[:First:];
+    } else {
+      return member_path<Rest...>::get(obj.[:First:]);
+    }
+  }
+};
+
+// True when T can be flattened: a structure deserialized member by member (not
+// a string, a container, an optional, a smart pointer, ...).
+template <typename T>
+constexpr bool flattenable_type = user_defined_type<T> && !concepts::string_view_keyed_map<T>
+    && !concepts::container_but_not_string<T> && !concepts::smart_pointer<T>;
+
+consteval void append_eligible_fields(std::meta::info type, std::vector<std::meta::info> &prefix,
+                                      std::vector<std::meta::info> &fields) {
+  for (std::meta::info mem : std::meta::nonstatic_data_members_of(type, std::meta::access_context::unchecked())) {
+    if (!is_eligible_member(mem)) { continue; }
+    prefix.push_back(std::meta::reflect_constant(mem));
+    if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::flatten_tag)) {
+      std::meta::info flattened = simdjson::detail::flattened_type(mem);
+      if (!std::meta::extract<bool>(std::meta::substitute(^^flattenable_type, {flattened}))) {
+        throw std::meta::exception(u8"simdjson::flatten requires a member whose type is a structure deserialized member by member", mem);
+      }
+      append_eligible_fields(flattened, prefix, fields);
+    } else {
+      fields.push_back(std::meta::substitute(^^member_path, prefix));
+    }
+    prefix.pop_back();
+  }
+}
+
+// The fields of `type` that participate in deserialization, in declaration order
+// (the fields of a flattened member take its place). A field's position in this
+// list is its "field index" below.
+consteval std::vector<std::meta::info> eligible_fields(std::meta::info type) {
+  std::vector<std::meta::info> prefix;
+  std::vector<std::meta::info> fields;
+  append_eligible_fields(type, prefix, fields);
+  return fields;
+}
+
+// The data member at the end of a field path.
+consteval std::meta::info field_leaf(std::meta::info path) {
+  return std::meta::extract<std::meta::info>(std::meta::template_arguments_of(path).back());
+}
+
+// Number of fields that participate in deserialization. A class can have zero
+// eligible fields (e.g. std::chrono::time_point, whose only data member is
+// private): an empty key_selector cannot be built, so the tag_invoke below
+// special-cases this count.
+template <typename T>
+consteval std::size_t eligible_field_count() {
+  return eligible_fields(^^T).size();
+}
+
+// The keys accepted for a member (or an enumerator): its JSON key followed by its
+// aliases. They are static strings, so they can be used as template arguments.
+consteval std::vector<const char *> accepted_keys_of(std::meta::info entity) {
+  std::vector<const char *> keys;
+  for (std::string_view key : simdjson::detail::json_key_names(entity)) {
+    bool repeated = false;
+    for (const char *previous : keys) { repeated = repeated || std::string_view(previous) == key; }
+    if (!repeated) { keys.push_back(std::define_static_string(key)); }
+  }
+  return keys;
+}
+
+// Every key accepted for T: the eligible fields in order, each followed by its
+// aliases. This is the key list of T's key_selector.
+consteval std::vector<const char *> accepted_keys(std::meta::info type) {
+  std::vector<const char *> keys;
+  for (std::meta::info path : eligible_fields(type)) {
+    for (const char *key : accepted_keys_of(field_leaf(path))) { keys.push_back(key); }
+  }
+  return keys;
+}
+
+// The field index of each entry of accepted_keys(type).
+consteval std::vector<std::size_t> accepted_key_fields(std::meta::info type) {
+  std::vector<std::size_t> key_fields;
+  std::vector<std::meta::info> fields = eligible_fields(type);
+  for (std::size_t i = 0; i < fields.size(); ++i) {
+    for (std::size_t j = 0; j < accepted_keys_of(field_leaf(fields[i])).size(); ++j) { key_fields.push_back(i); }
+  }
+  return key_fields;
+}
+
+// True when no two eligible fields of T accept the same key. Otherwise one JSON
+// key would have to fill several members: this is reported at compile time.
+template <typename T>
+consteval bool accepted_keys_are_distinct() {
+  std::vector<const char *> keys = accepted_keys(^^T);
+  for (std::size_t i = 0; i < keys.size(); ++i) {
+    for (std::size_t j = i + 1; j < keys.size(); ++j) {
+      if (std::string_view(keys[i]) == std::string_view(keys[j])) { return false; }
+    }
+  }
+  return true;
+}
+
+// True when some accepted key of T is written with escape sequences in JSON (a
+// double quote, a backslash or a control character). Such a key can only be
+// matched by comparing unescaped keys (deserialize_struct_scan): obj[key] and
+// the key_selector compare the raw bytes.
+template <typename T>
+consteval bool keys_need_unescaping() {
+  for (std::string_view key : accepted_keys(^^T)) {
+    for (char c : key) {
+      if (c == '\\' || c == '"' || static_cast<unsigned char>(c) < 0x20) { return true; }
+    }
+  }
+  return false;
+}

+// True when some eligible field of T has aliases: several selector keys may then
+// map to the same field.
+template <typename T>
+consteval bool has_aliases() {
+  return accepted_keys(^^T).size() != eligible_fields(^^T).size();
+}
+
+// True when `mem` is annotated with default_value or default_from (directly, or
+// through a default_value annotation on the enclosing structure).
+template <auto mem>
+consteval bool has_default() {
+  return simdjson::detail::has_annotation(mem, ^^simdjson::detail::default_value_tag)
+      || simdjson::detail::has_annotation(std::meta::parent_of(mem), ^^simdjson::detail::default_value_tag)
+      || simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::default_from_t) != std::meta::info{};
+}
+
+// True when a missing key for `mem` is not an error: optional members and
+// members with a default.
+template <auto mem>
+consteval bool may_be_absent() {
+  return concepts::optional_type<typename [: std::meta::type_of(mem) :]> || has_default<mem>();
+}
+
+// True when every eligible field of T is required (none may be absent). In that
+// case presence can be checked with a single match count instead of a per-field
+// "seen" array.
+template <typename T>
+consteval bool all_eligible_fields_required() {
+  bool all_required = true;
+  template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+    if constexpr (may_be_absent<[: path :]::leaf>()) {
+      all_required = false;
+    }
+  }
+  return all_required;
+}
+
+// `key` as a constevalutil::fixed_string usable as an NTTP.
+template <const char *key>
+consteval auto key_fixed_string() {
+  constexpr std::string_view key_view{ key };
+  char buffer[key_view.size() + 1] = {};
+  for (std::size_t i = 0; i < key_view.size(); ++i) { buffer[i] = key_view[i]; }
+  return constevalutil::fixed_string<key_view.size() + 1>(buffer);
+}
+
+// key_selector template arguments (one fixed_string per accepted key), in the
+// order of accepted_keys.
+template <typename T>
+consteval std::vector<std::meta::info> selector_key_args() {
+  std::vector<std::meta::info> args;
+  template for (constexpr const char *key : std::define_static_array(accepted_keys(^^T))) {
+    args.push_back(std::meta::reflect_constant(key_fixed_string<key>()));
+  }
+  return args;
+}
+
+// key_selector whose keys are exactly T's accepted keys (index i <-> i-th key).
+template <typename T>
+using selector_for = typename [: std::meta::substitute(
+    ^^ppc64::ondemand::key_selector, selector_key_args<T>()) :];
+
+// True when T's accepted keys satisfy every key_selector requirement, so a
+// key_selector can be built for T without a compile-time error. This mirrors the
+// key_selector limits (see key_selector.h): at most 255 keys, each key non-empty
+// and at most 63 characters, no backslash / double-quote / null byte, and all
+// keys distinct. Keys with any other control character are excluded too: they
+// are escaped in JSON, so their raw bytes never match. When this returns false
+// the deserializer falls back to deserialize_struct_scan instead of failing to
+// compile.
+template <typename T>
+consteval bool keys_fit_selector() {
+  std::vector<std::string_view> keys;
+  for (const char *key : accepted_keys(^^T)) { keys.push_back(key); }
+  if (keys.size() > 255) { return false; }
+  for (std::size_t i = 0; i < keys.size(); ++i) {
+    if (keys[i].empty() || keys[i].size() > 63) { return false; }
+    for (char c : keys[i]) {
+      if (c == '\\' || c == '"' || static_cast<unsigned char>(c) < 0x20) { return false; }
+    }
+    for (std::size_t j = i + 1; j < keys.size(); ++j) {
+      if (keys[i] == keys[j]) { return false; }
+    }
+  }
+  return true;
+}
+
+// True when the class `adapter` declares a member named deserialize.
+consteval bool declares_deserialize(std::meta::info adapter) {
+  for (std::meta::info m : std::meta::members_of(adapter, std::meta::access_context::unchecked())) {
+    if (std::meta::has_identifier(m) && std::meta::identifier_of(m) == "deserialize") { return true; }
+  }
+  return false;
+}
+
+// Deserialize a JSON value into `target`, the storage of member `mem`, through
+// the member's with<Adapter> annotation when the adapter provides a deserialize
+// function.
+template <auto mem, typename ValueT, typename M>
+simdjson_warn_unused simdjson_inline error_code deserialize_member_value(ValueT &field_value, M &target) noexcept(false) {
+  constexpr std::meta::info with_type = simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::with_t);
+  if constexpr (with_type != std::meta::info{}) {
+    using adapter = typename [: with_type :]::adapter;
+    using ondemand_value = ppc64::ondemand::value;
+    if constexpr (requires { { adapter::deserialize(field_value, target) } -> std::convertible_to<error_code>; }) {
+      return adapter::deserialize(field_value, target);
+    } else if constexpr (requires(ondemand_value &v) { { adapter::deserialize(v, target) } -> std::convertible_to<error_code>; }
+                         && requires { field_value.get_value(); }) {
+      // A transparent structure read from a document: the adapter takes an
+      // ondemand::value. A scalar document cannot be viewed as a value, so it
+      // reports SCALAR_DOCUMENT_AS_VALUE (an adapter taking auto& receives the
+      // document itself and has no such limitation).
+      ondemand_value v;
+      SIMDJSON_TRY(field_value.get_value().get(v));
+      return adapter::deserialize(v, target);
+    } else {
+      static_assert(!declares_deserialize(^^adapter),
+                    "the deserialize function of a simdjson::with adapter must be callable as "
+                    "Adapter::deserialize(simdjson::ondemand::value &, T &) and return an error_code");
+      return field_value.get(target);
+    }
+  } else {
+    return field_value.get(target);
+  }
+}
+
+// Deserialize the storage `target` of member `mem` from a JSON value.
+template <auto mem, typename ValueT, typename M>
+simdjson_warn_unused simdjson_inline error_code deserialize_member(ValueT &field_value, M &target) noexcept(false) {
+  if constexpr (has_default<mem>() && std::is_default_constructible_v<M> && std::is_move_assignable_v<M>) {
+    // A present key replaces the default value: deserialize into a fresh
+    // temporary so that, e.g., a container does not append to its default
+    // content, and a failure leaves the default untouched.
+    M value{};
+    SIMDJSON_TRY(deserialize_member_value<mem>(field_value, value));
+    target = std::move(value);
+    return SUCCESS;
+  } else {
+    return deserialize_member_value<mem>(field_value, target);
+  }
+}

+// Deserialize the field with the given field index.
+template <typename T>
+simdjson_warn_unused simdjson_inline error_code deserialize_field_at(
+    std::size_t field_index, ppc64::ondemand::value field_value, T &out) noexcept(false) {
+  std::size_t counter = 0;
+  template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+    using field = [: path :];
+    if (field_index == counter) { return deserialize_member<field::leaf>(field_value, field::get(out)); }
+    ++counter;
+  }
+  return SUCCESS;
+}
+
+// Called when the key(s) of member `mem`, stored in `target`, are missing from
+// the JSON object: an error for a required member, a call to the default_from
+// factory, or nothing at all.
+template <auto mem, typename M>
+simdjson_warn_unused simdjson_inline error_code on_missing_member(M &target) noexcept(false) {
+  constexpr std::meta::info default_from_type = simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::default_from_t);
+  if constexpr (default_from_type != std::meta::info{}) {
+    target = [: default_from_type :]::factory();
+    return SUCCESS;
+  } else if constexpr (may_be_absent<mem>()) {
+    // For optional and default_value members, a missing key is not an error:
+    // leave the member at its current (default) value.
+    (void)target;
+    return SUCCESS;
+  } else {
+    (void)target;
+    return NO_SUCH_FIELD;
+  }
+}
+
+// Report or handle every field that was not seen in the JSON object.
+template <typename T, std::size_t N>
+simdjson_warn_unused simdjson_inline error_code handle_missing_fields(
+    const std::array<bool, N> &seen_field, T &out) noexcept(false) {
+  std::size_t counter = 0;
+  template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+    using field = [: path :];
+    if (!seen_field[counter]) { SIMDJSON_TRY(on_missing_member<field::leaf>(field::get(out))); }
+    ++counter;
+  }
+  return SUCCESS;
+}
+
+// Ordered, per-field deserialization: one obj[key] lookup per eligible field
+// (and per alias, until one is found). This is the opt-out path
+// (-DSIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION=1), except for structs with keys
+// that need unescaping (see keys_need_unescaping).
+template <typename T>
+simdjson_warn_unused error_code deserialize_struct_ordered(
+    ppc64::ondemand::object &obj, T &out) noexcept(false) {
+  template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+    using field = [: path :];
+    ppc64::ondemand::value field_value;
+    error_code error = NO_SUCH_FIELD;
+    template for (constexpr const char *key : std::define_static_array(accepted_keys_of(field::leaf))) {
+      if (error == NO_SUCH_FIELD) { error = obj[std::string_view(key)].get(field_value); }
+    }
+    if (error == NO_SUCH_FIELD) {
+      SIMDJSON_TRY(on_missing_member<field::leaf>(field::get(out)));
+    } else if (error) {
+      return error;
+    } else {
+      SIMDJSON_TRY(deserialize_member<field::leaf>(field_value, field::get(out)));
+    }
+  }
+  return SUCCESS;
+}
+
+// Appends the JSON keys that serialization writes for members of `type` that
+// deserialization cannot assign (const or non-public members, directly or
+// through flatten). `all` is true inside a flattened member that is itself
+// unassignable. Keys of skip_deserializing members are not included: they are
+// unknown keys, as documented.
+consteval void append_unassignable_keys(std::meta::info type, bool all, std::vector<const char *> &keys) {
+  for (std::meta::info mem : std::meta::nonstatic_data_members_of(type, std::meta::access_context::unchecked())) {
+    if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_tag)
+        || simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_serializing_tag)
+        || simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_deserializing_tag)) {
+      continue;
+    }
+    bool unassignable = all || !is_eligible_member(mem);
+    if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::flatten_tag)) {
+      append_unassignable_keys(simdjson::detail::flattened_type(mem), unassignable, keys);
+    } else if (unassignable) {
+      keys.push_back(std::define_static_string(simdjson::detail::json_key_name(mem)));
+    }
+  }
+}
+
+consteval std::vector<const char *> unassignable_keys(std::meta::info type) {
+  std::vector<const char *> keys;
+  append_unassignable_keys(type, false, keys);
+  return keys;
+}
+
+// Deserialization by a single pass over every field of the object, comparing
+// unescaped keys. It is used for structs annotated with deny_unknown_fields
+// (DenyUnknown = true), where a key that does not map to an eligible field is
+// reported as UNKNOWN_FIELD, and as the fallback for structs whose keys the
+// key_selector or obj[key] cannot match (see keys_fit_selector and
+// keys_need_unescaping). As with object::for_each, the first occurrence of a
+// field wins (later duplicates, or aliases of a field already seen, are ignored).
+template <bool DenyUnknown, typename T>
+simdjson_warn_unused error_code deserialize_struct_scan(
+    ppc64::ondemand::object &obj, T &out) noexcept(false) {
+  static constexpr auto keys = std::define_static_array(accepted_keys(^^T));
+  static constexpr auto key_fields = std::define_static_array(accepted_key_fields(^^T));
+  std::array<bool, eligible_field_count<T>()> seen_field{};
+  for (auto field_result : obj) {
+    ppc64::ondemand::field json_field;
+    SIMDJSON_TRY(std::move(field_result).get(json_field));
+    std::string_view key;
+    SIMDJSON_TRY(json_field.unescaped_key().get(key));
+    std::size_t key_index = keys.size();
+    for (std::size_t i = 0; i < keys.size(); ++i) {
+      if (key == std::string_view(keys[i])) { key_index = i; break; }
+    }
+    if (key_index == keys.size()) {
+      if constexpr (DenyUnknown) {
+        // A key that T itself serializes (e.g. of a const member) is not
+        // unknown: a serialized value must parse back.
+        static constexpr auto ignored_keys = std::define_static_array(unassignable_keys(^^T));
+        bool ignored = false;
+        for (const char *ignored_key : ignored_keys) {
+          if (key == std::string_view(ignored_key)) { ignored = true; break; }
+        }
+        if (!ignored) { return UNKNOWN_FIELD; }
+      }
+      continue;
+    }
+    const std::size_t field_index = key_fields[key_index];
+    if (seen_field[field_index]) { continue; }
+    seen_field[field_index] = true;
+    SIMDJSON_TRY(deserialize_field_at(field_index, json_field.value(), out));
+  }
+  return handle_missing_fields(seen_field, out);
+}
+
+template <typename T>
+consteval bool is_transparent() {
+  return simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag);
+}
+
+} // namespace key_selector_reflection_detail
+
+// Deserialize a reflected struct. By default this builds a compile-time
+// key_selector from the struct's members and walks each object once with
+// object::for_each (perfect-hash key matching), instead of one obj[key] lookup
+// per member. Other paths are used instead:
+//   - globally, the ordered per-member path when defining
+//     -DSIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION=1;
+//   - automatically and per-type, a scan of the object comparing unescaped keys
+//     when the struct's keys do not fit the key_selector limits (see
+//     keys_fit_selector) or cannot be compared raw (see keys_need_unescaping),
+//     so that long member names and the like keep compiling rather than
+//     tripping a static_assert.
+// Structs annotated with deny_unknown_fields always use the strict scan, and
+// structs annotated with transparent are deserialized as their single member.
+//
+// noexcept(false): a member's tag_invoke may throw and the exception must reach
+// the caller of get<T>().
 template <typename T, typename ValT>
   requires(user_defined_type<T> && std::is_class_v<T>)
-error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
+  if constexpr (key_selector_reflection_detail::is_transparent<T>()) {
+    constexpr auto mem = simdjson::detail::transparent_member(^^T);
+    if constexpr (std::is_same_v<std::remove_cvref_t<ValT>, ppc64::ondemand::object>) {
+      // We were handed an object: only a structure can be deserialized from it.
+      if constexpr (user_defined_type<typename [: std::meta::type_of(mem) :]>) {
+        return tag_invoke(deserialize_tag{}, val, out.[:mem:]);
+      } else {
+        return INCORRECT_TYPE;
+      }
+    } else {
+      return key_selector_reflection_detail::deserialize_member<mem>(val, out.[:mem:]);
+    }
+  } else {
+  static_assert(key_selector_reflection_detail::accepted_keys_are_distinct<T>(),
+                "two members of this structure accept the same JSON key (check rename, alias, "
+                "rename_all and flatten)");
   ppc64::ondemand::object obj;
   if constexpr (std::is_same_v<std::remove_cvref_t<ValT>, ppc64::ondemand::object>) {
     obj = val;
   } else {
     SIMDJSON_TRY(val.get_object().get(obj));
   }
-  template for (constexpr auto mem : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
-    if constexpr (!std::meta::is_const(mem) && std::meta::is_public(mem)) {
-      constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
-      if constexpr (concepts::optional_type<decltype(out.[:mem:])>) {
-        // for optional members, it's ok if the key is missing
-        auto error = obj[key].get(out.[:mem:]);
-        if (error && error != NO_SUCH_FIELD) {
-          if(error == NO_SUCH_FIELD) {
-            out.[:mem:].reset();
-            continue;
-          }
-          return error;
-        }
-      } else {
-        // for non-optional members, the key must be present
-        SIMDJSON_TRY(obj[key].get(out.[:mem:]));
+  if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::deny_unknown_fields_tag)) {
+    return key_selector_reflection_detail::deserialize_struct_scan<true>(obj, out);
+  } else {
+#if defined(SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION) && SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION
+  // Opt-out build: use the ordered per-member path, unless obj[key] cannot
+  // match T's keys.
+  if constexpr (key_selector_reflection_detail::keys_need_unescaping<T>()) {
+    return key_selector_reflection_detail::deserialize_struct_scan<false>(obj, out);
+  } else {
+    return key_selector_reflection_detail::deserialize_struct_ordered(obj, out);
+  }
+#else
+  if constexpr (key_selector_reflection_detail::eligible_field_count<T>() == 0) {
+    // No fields to deserialize: an empty key_selector cannot be built, so just
+    // validate that the input is an object (done above) and succeed. Mirrors the
+    // ordered per-member path, which iterates over zero members.
+    (void)out;
+    (void)obj;
+    return SUCCESS;
+  } else if constexpr (!key_selector_reflection_detail::keys_fit_selector<T>()) {
+    // Automatic fallback: T's accepted keys do not fit the key_selector limits
+    // (e.g. a member name longer than 63 characters, or a key with a double
+    // quote), so building a selector would be a compile error. Scan the object
+    // instead, so the default never breaks a struct that the opt-out path would
+    // accept.
+    return key_selector_reflection_detail::deserialize_struct_scan<false>(obj, out);
+  } else {
+  using selector = key_selector_reflection_detail::selector_for<T>;
+  if constexpr (key_selector_reflection_detail::all_eligible_fields_required<T>()
+                && !key_selector_reflection_detail::has_aliases<T>()) {
+    // Fast path: every member is required and has a single key. A single
+    // for_each pass parses each matched field; the returned match count then
+    // tells us whether every member was present (matched_count ==
+    // selector::size()) without a per-member "seen" array. A value-parse error
+    // (e.g. a type mismatch) is propagated by for_each.
+    auto walk = obj.template for_each<selector>(
+        [&](std::size_t matched_index, ppc64::ondemand::value field_value) -> error_code {
+      std::size_t counter = 0;
+      template for (constexpr auto path : std::define_static_array(key_selector_reflection_detail::eligible_fields(^^T))) {
+        using field = [: path :];
+        if (matched_index == counter) { return key_selector_reflection_detail::deserialize_member<field::leaf>(field_value, field::get(out)); }
+        ++counter;
       }
-    }
-  };
-  return simdjson::SUCCESS;
-}
-
-// Support for enum deserialization - deserialize from string representation using expand approach from P2996R12
+      return SUCCESS;
+    });
+    if (walk.error) { return walk.error; }
+    // A missing required member shows up as a short match count and is reported as
+    // NO_SUCH_FIELD, mirroring the ordered obj[key] path.
+    if (walk.matched_count != selector::size()) { return NO_SUCH_FIELD; }
+    return SUCCESS;
+  } else {
+    static constexpr auto key_fields = std::define_static_array(key_selector_reflection_detail::accepted_key_fields(^^T));
+    std::array<bool, key_selector_reflection_detail::eligible_field_count<T>()> seen_field{};
+    // Single pass over the object: each field whose key matches a member (or one
+    // of its aliases) yields its selector index, which we map back to the
+    // corresponding member. The first key seen for a member wins. The callback
+    // returns an error_code so that a value-parse error (e.g. a type mismatch on
+    // a matched field) is propagated by for_each instead of being silently dropped.
+    error_code walk_error = obj.template for_each<selector>(
+        [&](std::size_t matched_index, ppc64::ondemand::value field_value) -> error_code {
+      const std::size_t field_index = key_fields[matched_index];
+      if (seen_field[field_index]) { return SUCCESS; }
+      seen_field[field_index] = true;
+      return key_selector_reflection_detail::deserialize_field_at(field_index, field_value, out);
+    });
+    if (walk_error) { return walk_error; }
+    // Required members must be present: a missing one is reported as
+    // NO_SUCH_FIELD, mirroring the ordered obj[key] path. Optional and defaulted
+    // members may be absent.
+    return key_selector_reflection_detail::handle_missing_fields(seen_field, out);
+  }
+  }
+#endif // SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION
+  }
+  }
+}
+
+// Support for enum deserialization - deserialize from string representation using
+// expand approach from P2996R12. The accepted strings are the enumerator's JSON key
+// (see rename and rename_all) and its aliases.
 template <typename T, typename ValT>
   requires(std::is_enum_v<T>)
 error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
 #if SIMDJSON_STATIC_REFLECTION
   std::string_view str;
   SIMDJSON_TRY(val.get_string().get(str));
-  constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
+  static constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
   template for (constexpr auto enum_val : enumerators) {
-    if (str == std::meta::identifier_of(enum_val)) {
-      out = [:enum_val:];
-      return SUCCESS;
+    template for (constexpr const char *key : std::define_static_array(key_selector_reflection_detail::accepted_keys_of(enum_val))) {
+      if (str == std::string_view(key)) {
+        out = [:enum_val:];
+        return SUCCESS;
+      }
     }
   };

@@ -123445,33 +157740,25 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {

 template <typename simdjson_value, typename T>
   requires(user_defined_type<std::remove_cvref_t<T>>)
-error_code tag_invoke(deserialize_tag, simdjson_value &val, std::unique_ptr<T> &out) noexcept {
-  if (!out) {
-    out = std::make_unique<T>();
-    if (!out) {
-      return MEMALLOC;
-    }
-  }
-  if (auto err = val.get(*out)) {
-    out.reset();
-    return err;
+error_code tag_invoke(deserialize_tag, simdjson_value &val, std::unique_ptr<T> &out) noexcept(false) {
+  std::unique_ptr<T> ptr(new (std::nothrow) T());
+  if (!ptr) {
+    return MEMALLOC;
   }
+  SIMDJSON_TRY(val.get(*ptr));
+  out = std::move(ptr);
   return SUCCESS;
 }

 template <typename simdjson_value, typename T>
   requires(user_defined_type<std::remove_cvref_t<T>>)
-error_code tag_invoke(deserialize_tag, simdjson_value &val, std::shared_ptr<T> &out) noexcept {
-  if (!out) {
-    out = std::make_shared<T>();
-    if (!out) {
-      return MEMALLOC;
-    }
-  }
-  if (auto err = val.get(*out)) {
-    out.reset();
-    return err;
+error_code tag_invoke(deserialize_tag, simdjson_value &val, std::shared_ptr<T> &out) noexcept(false) {
+  std::shared_ptr<T> ptr(new (std::nothrow) T());
+  if (!ptr) {
+    return MEMALLOC;
   }
+  SIMDJSON_TRY(val.get(*ptr));
+  out = std::move(ptr);
   return SUCCESS;
 }

@@ -123783,9 +158070,17 @@ simdjson_inline simdjson_result<array> array::started(value_iterator &iter) noex
   return array(iter);
 }

-simdjson_inline simdjson_result<array_iterator> array::begin() noexcept {
+simdjson_inline simdjson_result<array_iterator> array::begin() & noexcept {
 #if SIMDJSON_DEVELOPMENT_CHECKS
-  if (!iter.is_at_iterator_start()) { return OUT_OF_ORDER_ITERATION; }
+  if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
+  return array_iterator(iter, this);
+#endif
+  return array_iterator(iter);
+}
+simdjson_inline simdjson_result<array_iterator> array::begin() && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  // The array is a temporary that the iterator may outlive: do not lock it.
+  if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
 #endif
   return array_iterator(iter);
 }
@@ -123812,6 +158107,9 @@ simdjson_inline simdjson_result<std::string_view> array::raw_json() noexcept {
 SIMDJSON_PUSH_DISABLE_WARNINGS
 SIMDJSON_DISABLE_STRICT_OVERFLOW_WARNING
 simdjson_inline simdjson_result<size_t> array::count_elements() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   size_t count{0};
   // Important: we do not consume any of the values.
   for(simdjson_unused auto v : *this) { count++; }
@@ -123825,6 +158123,9 @@ simdjson_inline simdjson_result<size_t> array::count_elements() & noexcept {
 SIMDJSON_POP_DISABLE_WARNINGS

 simdjson_inline simdjson_result<bool> array::is_empty() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   bool is_not_empty;
   auto error = iter.reset_array().get(is_not_empty);
   if(error) { return error; }
@@ -123832,31 +158133,30 @@ simdjson_inline simdjson_result<bool> array::is_empty() & noexcept {
 }

 inline simdjson_result<bool> array::reset() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   return iter.reset_array();
 }

+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline void array::set_locked(bool _locked) noexcept {
+  locked = _locked;
+}
+#endif
+
 inline simdjson_result<value> array::at_pointer(std::string_view json_pointer) noexcept {
-  if (json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
+  // An empty pointer has no json_pointer[0]: with a default-constructed
+  // std::string_view, reading it dereferences a null pointer.
+  if (json_pointer.empty() || json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
   json_pointer = json_pointer.substr(1);
   // - means "the append position" or "the element after the end of the array"
   // We don't support this, because we're returning a real element, not a position.
   if (json_pointer == "-") { return INDEX_OUT_OF_BOUNDS; }

-  // Read the array index
   size_t array_index = 0;
   size_t i;
-  for (i = 0; i < json_pointer.length() && json_pointer[i] != '/'; i++) {
-    uint8_t digit = uint8_t(json_pointer[i] - '0');
-    // Check for non-digit in array index. If it's there, we're trying to get a field in an object
-    if (digit > 9) { return INCORRECT_TYPE; }
-    array_index = array_index*10 + digit;
-  }
-
-  // 0 followed by other digits is invalid
-  if (i > 1 && json_pointer[0] == '0') { return INVALID_JSON_POINTER; } // "JSON pointer array index has other characters after 0"
-
-  // Empty string is invalid; so is a "/" with no digits before it
-  if (i == 0) { return INVALID_JSON_POINTER; } // "Empty string in JSON pointer array index"
+  SIMDJSON_TRY(internal::parse_json_pointer_array_index(json_pointer, array_index, i));
   // Get the child
   auto child = at(array_index);
   // If there is an error, it ends here
@@ -123930,6 +158230,9 @@ inline error_code array::for_each_at_path_with_wildcard(std::string_view json_pa
 }

 simdjson_inline simdjson_result<value> array::at(size_t index) noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   size_t i = 0;
   for (auto value : *this) {
     if (i == index) { return value; }
@@ -123959,10 +158262,14 @@ simdjson_inline simdjson_result<ppc64::ondemand::array>::simdjson_result(
 {
 }

-simdjson_inline simdjson_result<ppc64::ondemand::array_iterator> simdjson_result<ppc64::ondemand::array>::begin() noexcept {
+simdjson_inline simdjson_result<ppc64::ondemand::array_iterator> simdjson_result<ppc64::ondemand::array>::begin() & noexcept {
   if (error()) { return error(); }
   return first.begin();
 }
+simdjson_inline simdjson_result<ppc64::ondemand::array_iterator> simdjson_result<ppc64::ondemand::array>::begin() && noexcept {
+  if (error()) { return error(); }
+  return std::move(first).begin();
+}
 simdjson_inline simdjson_result<ppc64::ondemand::array_iterator> simdjson_result<ppc64::ondemand::array>::end() noexcept {
   if (error()) { return error(); }
   return first.end();
@@ -124025,6 +158332,59 @@ simdjson_inline array_iterator::array_iterator(const value_iterator &_iter) noex
   : iter{_iter}
 {}

+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline array_iterator::array_iterator(const value_iterator &_iter, array* _parent) noexcept
+  : parent{_parent}, iter{_iter}
+{
+  if (parent) parent->set_locked(true);
+}
+
+simdjson_inline array_iterator::~array_iterator() noexcept
+{
+  if (parent) parent->set_locked(false);
+}
+
+simdjson_inline array_iterator::array_iterator(array_iterator&& other) noexcept
+  : has_been_referenced{other.has_been_referenced},
+    parent{other.parent},
+    iter{std::move(other.iter)}
+{
+  other.parent = nullptr;
+}
+
+simdjson_inline array_iterator& array_iterator::operator=(array_iterator&& other) noexcept {
+  if (this != &other)
+  {
+    if (parent)
+      parent->set_locked(false);
+    has_been_referenced = other.has_been_referenced;
+    parent = other.parent;
+    iter = std::move(other.iter);
+
+    other.parent = nullptr;
+  }
+  return *this;
+}
+
+simdjson_inline array_iterator::array_iterator(const array_iterator& other) noexcept
+  : has_been_referenced{other.has_been_referenced},
+    parent{nullptr},
+    iter{other.iter}
+{}
+
+simdjson_inline array_iterator& array_iterator::operator=(const array_iterator& other) noexcept {
+  if (this != &other)
+  {
+    if (parent)
+      parent->set_locked(false);
+    has_been_referenced = other.has_been_referenced;
+    parent = nullptr;
+    iter = other.iter;
+  }
+  return *this;
+}
+#endif
+
 simdjson_inline simdjson_result<value> array_iterator::operator*() noexcept {
 #if SIMDJSON_DEVELOPMENT_CHECKS
    SIMDJSON_ASSUME(!has_been_referenced);
@@ -124120,6 +158480,41 @@ namespace simdjson {
 namespace ppc64 {
 namespace ondemand {

+/**
+ * @private Narrows the result of get_uint64() to a smaller unsigned integer type T.
+ * @returns NUMBER_OUT_OF_RANGE if the value does not fit in a T.
+ */
+template <typename T>
+simdjson_inline simdjson_result<T> narrow_integer(simdjson_result<uint64_t> wide) noexcept {
+  uint64_t result;
+  SIMDJSON_TRY(std::move(wide).get(result));
+  if (result > static_cast<uint64_t>((std::numeric_limits<T>::max)())) { return NUMBER_OUT_OF_RANGE; }
+  return static_cast<T>(result);
+}
+
+/**
+ * @private Narrows the result of get_int64() to a smaller signed integer type T.
+ * @returns NUMBER_OUT_OF_RANGE if the value does not fit in a T.
+ */
+template <typename T>
+simdjson_inline simdjson_result<T> narrow_integer(simdjson_result<int64_t> wide) noexcept {
+  int64_t result;
+  SIMDJSON_TRY(std::move(wide).get(result));
+  if (result > static_cast<int64_t>((std::numeric_limits<T>::max)()) || result < static_cast<int64_t>((std::numeric_limits<T>::min)())) { return NUMBER_OUT_OF_RANGE; }
+  return static_cast<T>(result);
+}
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+// get_float32() parses with get_float(), so float must be binary32.
+static_assert(std::numeric_limits<float>::is_iec559 && std::numeric_limits<float>::digits == 24,
+              "get_float32() requires float to be IEEE binary32");
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+// get_float64() parses with get_double(), so double must be binary64.
+static_assert(std::numeric_limits<double>::is_iec559 && std::numeric_limits<double>::digits == 53,
+              "get_float64() requires double to be IEEE binary64");
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
+
 simdjson_inline value::value(const value_iterator &_iter) noexcept
   : iter{_iter}
 {
@@ -124151,6 +158546,13 @@ simdjson_inline simdjson_result<raw_json_string> value::get_raw_json_string() no
 simdjson_inline simdjson_result<std::string_view> value::get_string(bool allow_replacement) noexcept {
   return iter.get_string(allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> value::get_u8string(bool allow_replacement) noexcept {
+  std::string_view content;
+  SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+  return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code value::get_string(string_type& receiver, bool allow_replacement) noexcept {
   return iter.get_string(receiver, allow_replacement);
@@ -124164,6 +158566,12 @@ simdjson_inline simdjson_result<double> value::get_double() noexcept {
 simdjson_inline simdjson_result<double> value::get_double_in_string() noexcept {
   return iter.get_double_in_string();
 }
+simdjson_inline simdjson_result<float> value::get_float() noexcept {
+  return iter.get_float();
+}
+simdjson_inline simdjson_result<float> value::get_float_in_string() noexcept {
+  return iter.get_float_in_string();
+}
 simdjson_inline simdjson_result<uint64_t> value::get_uint64() noexcept {
   return iter.get_uint64();
 }
@@ -124177,17 +158585,37 @@ simdjson_inline simdjson_result<int64_t> value::get_int64_in_string() noexcept {
   return iter.get_int64_in_string();
 }
 simdjson_inline simdjson_result<uint32_t> value::get_uint32() noexcept {
-  uint64_t result;
-  SIMDJSON_TRY(get_uint64().get(result));
-  if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<uint32_t>(result);
+  return narrow_integer<uint32_t>(get_uint64());
 }
 simdjson_inline simdjson_result<int32_t> value::get_int32() noexcept {
-  int64_t result;
-  SIMDJSON_TRY(get_int64().get(result));
-  if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<int32_t>(result);
+  return narrow_integer<int32_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint16_t> value::get_uint16() noexcept {
+  return narrow_integer<uint16_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int16_t> value::get_int16() noexcept {
+  return narrow_integer<int16_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint8_t> value::get_uint8() noexcept {
+  return narrow_integer<uint8_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int8_t> value::get_int8() noexcept {
+  return narrow_integer<int8_t>(get_int64());
 }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> value::get_float32() noexcept {
+  float result;
+  SIMDJSON_TRY(get_float().get(result));
+  return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> value::get_float64() noexcept {
+  double result;
+  SIMDJSON_TRY(get_double().get(result));
+  return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<bool> value::get_bool() noexcept {
   return iter.get_bool();
 }
@@ -124199,12 +158627,26 @@ template<> simdjson_inline simdjson_result<array> value::get() noexcept { return
 template<> simdjson_inline simdjson_result<object> value::get() noexcept { return get_object(); }
 template<> simdjson_inline simdjson_result<raw_json_string> value::get() noexcept { return get_raw_json_string(); }
 template<> simdjson_inline simdjson_result<std::string_view> value::get() noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> value::get() noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_inline simdjson_result<number> value::get() noexcept { return get_number(); }
 template<> simdjson_inline simdjson_result<double> value::get() noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> value::get() noexcept { return get_float(); }
 template<> simdjson_inline simdjson_result<uint64_t> value::get() noexcept { return get_uint64(); }
 template<> simdjson_inline simdjson_result<int64_t> value::get() noexcept { return get_int64(); }
 template<> simdjson_inline simdjson_result<uint32_t> value::get() noexcept { return get_uint32(); }
 template<> simdjson_inline simdjson_result<int32_t> value::get() noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> value::get() noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> value::get() noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> value::get() noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> value::get() noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> value::get() noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> value::get() noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<> simdjson_inline simdjson_result<bool> value::get() noexcept { return get_bool(); }


@@ -124212,12 +158654,26 @@ template<> simdjson_warn_unused simdjson_inline error_code value::get(array& out
 template<> simdjson_warn_unused simdjson_inline error_code value::get(object& out) noexcept { return get_object().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(raw_json_string& out) noexcept { return get_raw_json_string().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(std::string_view& out) noexcept { return get_string(false).get(out); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::u8string_view& out) noexcept { return get_u8string(false).get(out); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_warn_unused simdjson_inline error_code value::get(number& out) noexcept { return get_number().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(double& out) noexcept { return get_double().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(float& out) noexcept { return get_float().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(uint64_t& out) noexcept { return get_uint64().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(int64_t& out) noexcept { return get_int64().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(uint32_t& out) noexcept { return get_uint32().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(int32_t& out) noexcept { return get_int32().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(uint16_t& out) noexcept { return get_uint16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(int16_t& out) noexcept { return get_int16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(uint8_t& out) noexcept { return get_uint8().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(int8_t& out) noexcept { return get_int8().get(out); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::float32_t& out) noexcept { return get_float32().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::float64_t& out) noexcept { return get_float64().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<>  simdjson_warn_unused simdjson_inline error_code value::get(bool& out) noexcept { return get_bool().get(out); }

 #if SIMDJSON_EXCEPTIONS
@@ -124386,6 +158842,9 @@ inline bool is_pointer_well_formed(std::string_view json_pointer) noexcept {
 }

 simdjson_inline simdjson_result<value> value::at_pointer(std::string_view json_pointer) noexcept {
+  // The empty JSON Pointer refers to the whole value (RFC 6901), as in
+  // document::at_pointer.
+  if (json_pointer.empty()) { return value(iter); }
   json_type t;
   SIMDJSON_TRY(type().get(t));
   switch (t)
@@ -124423,6 +158882,10 @@ template <typename Func>
 template <typename Func>
 #endif
 inline error_code value::for_each_at_path_with_wildcard(std::string_view json_path, Func&& callback) noexcept {
+  // Every recursive step of for_each_at_path_with_wildcard goes through this
+  // function, and each one descends one level into the document. A path with
+  // many segments applied to a deeply nested document would otherwise recurse
+  // without bound and overflow the stack.
   if (size_t(iter.depth()) >= iter.json_iter().parser->max_depth()) { return DEPTH_ERROR; }
   json_type t;
   SIMDJSON_TRY(type().get(t));
@@ -124536,10 +158999,46 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<ppc64::ondemand::value>
   if (error()) { return error(); }
   return first.get_int32();
 }
+simdjson_inline simdjson_result<uint16_t> simdjson_result<ppc64::ondemand::value>::get_uint16() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<ppc64::ondemand::value>::get_int16() noexcept {
+  if (error()) { return error(); }
+  return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<ppc64::ondemand::value>::get_uint8() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<ppc64::ondemand::value>::get_int8() noexcept {
+  if (error()) { return error(); }
+  return first.get_int8();
+}
 simdjson_inline simdjson_result<double> simdjson_result<ppc64::ondemand::value>::get_double() noexcept {
   if (error()) { return error(); }
   return first.get_double();
 }
+simdjson_inline simdjson_result<float> simdjson_result<ppc64::ondemand::value>::get_float() noexcept {
+  if (error()) { return error(); }
+  return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<ppc64::ondemand::value>::get_float_in_string() noexcept {
+  if (error()) { return error(); }
+  return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<ppc64::ondemand::value>::get_float32() noexcept {
+  if (error()) { return error(); }
+  return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<ppc64::ondemand::value>::get_float64() noexcept {
+  if (error()) { return error(); }
+  return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<double> simdjson_result<ppc64::ondemand::value>::get_double_in_string() noexcept {
   if (error()) { return error(); }
   return first.get_double_in_string();
@@ -124548,6 +159047,12 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<ppc64::ondeman
   if (error()) { return error(); }
   return first.get_string(allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<ppc64::ondemand::value>::get_u8string(bool allow_replacement) noexcept {
+  if (error()) { return error(); }
+  return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_inline error_code simdjson_result<ppc64::ondemand::value>::get_string(string_type& receiver, bool allow_replacement) noexcept {
   if (error()) { return error(); }
@@ -124576,11 +159081,23 @@ template<> simdjson_inline error_code simdjson_result<ppc64::ondemand::value>::g
   return SUCCESS;
 }

-template<typename T> simdjson_inline simdjson_result<T> simdjson_result<ppc64::ondemand::value>::get() noexcept {
+template<typename T> simdjson_inline simdjson_result<T> simdjson_result<ppc64::ondemand::value>::get()
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, ppc64::ondemand::value>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>();
 }
-template<typename T> simdjson_inline error_code simdjson_result<ppc64::ondemand::value>::get(T &out) noexcept {
+template<typename T> simdjson_inline error_code simdjson_result<ppc64::ondemand::value>::get(T &out)
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, ppc64::ondemand::value>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>(out);
 }
@@ -124850,16 +159367,22 @@ simdjson_inline simdjson_result<int64_t> document::get_int64_in_string() noexcep
   return get_root_value_iterator().get_root_int64_in_string(true);
 }
 simdjson_inline simdjson_result<uint32_t> document::get_uint32() noexcept {
-  uint64_t result;
-  SIMDJSON_TRY(get_uint64().get(result));
-  if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<uint32_t>(result);
+  return narrow_integer<uint32_t>(get_uint64());
 }
 simdjson_inline simdjson_result<int32_t> document::get_int32() noexcept {
-  int64_t result;
-  SIMDJSON_TRY(get_int64().get(result));
-  if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<int32_t>(result);
+  return narrow_integer<int32_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint16_t> document::get_uint16() noexcept {
+  return narrow_integer<uint16_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int16_t> document::get_int16() noexcept {
+  return narrow_integer<int16_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint8_t> document::get_uint8() noexcept {
+  return narrow_integer<uint8_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int8_t> document::get_int8() noexcept {
+  return narrow_integer<int8_t>(get_int64());
 }
 simdjson_inline simdjson_result<double> document::get_double() noexcept {
   return get_root_value_iterator().get_root_double(true);
@@ -124867,9 +159390,36 @@ simdjson_inline simdjson_result<double> document::get_double() noexcept {
 simdjson_inline simdjson_result<double> document::get_double_in_string() noexcept {
   return get_root_value_iterator().get_root_double_in_string(true);
 }
+simdjson_inline simdjson_result<float> document::get_float() noexcept {
+  return get_root_value_iterator().get_root_float(true);
+}
+simdjson_inline simdjson_result<float> document::get_float_in_string() noexcept {
+  return get_root_value_iterator().get_root_float_in_string(true);
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> document::get_float32() noexcept {
+  float result;
+  SIMDJSON_TRY(get_float().get(result));
+  return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> document::get_float64() noexcept {
+  double result;
+  SIMDJSON_TRY(get_double().get(result));
+  return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<std::string_view> document::get_string(bool allow_replacement) noexcept {
   return get_root_value_iterator().get_root_string(true, allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> document::get_u8string(bool allow_replacement) noexcept {
+  std::string_view content;
+  SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+  return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code document::get_string(string_type& receiver, bool allow_replacement) noexcept {
   return get_root_value_iterator().get_root_string(receiver, true, allow_replacement);
@@ -124891,11 +159441,25 @@ template<> simdjson_inline simdjson_result<array> document::get() & noexcept { r
 template<> simdjson_inline simdjson_result<object> document::get() & noexcept { return get_object(); }
 template<> simdjson_inline simdjson_result<raw_json_string> document::get() & noexcept { return get_raw_json_string(); }
 template<> simdjson_inline simdjson_result<std::string_view> document::get() & noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> document::get() & noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_inline simdjson_result<double> document::get() & noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> document::get() & noexcept { return get_float(); }
 template<> simdjson_inline simdjson_result<uint64_t> document::get() & noexcept { return get_uint64(); }
 template<> simdjson_inline simdjson_result<int64_t> document::get() & noexcept { return get_int64(); }
 template<> simdjson_inline simdjson_result<uint32_t> document::get() & noexcept { return get_uint32(); }
 template<> simdjson_inline simdjson_result<int32_t> document::get() & noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> document::get() & noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> document::get() & noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> document::get() & noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> document::get() & noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> document::get() & noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> document::get() & noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<> simdjson_inline simdjson_result<bool> document::get() & noexcept { return get_bool(); }
 template<> simdjson_inline simdjson_result<value> document::get() & noexcept { return get_value(); }

@@ -124903,17 +159467,35 @@ template<> simdjson_warn_unused simdjson_inline error_code document::get(array&
 template<> simdjson_warn_unused simdjson_inline error_code document::get(object& out) & noexcept { return get_object().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(raw_json_string& out) & noexcept { return get_raw_json_string().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(std::string_view& out) & noexcept { return get_string(false).get(out); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::u8string_view& out) & noexcept { return get_u8string(false).get(out); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_warn_unused simdjson_inline error_code document::get(double& out) & noexcept { return get_double().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(float& out) & noexcept { return get_float().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(uint64_t& out) & noexcept { return get_uint64().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(int64_t& out) & noexcept { return get_int64().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(uint32_t& out) & noexcept { return get_uint32().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(int32_t& out) & noexcept { return get_int32().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(uint16_t& out) & noexcept { return get_uint16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(int16_t& out) & noexcept { return get_int16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(uint8_t& out) & noexcept { return get_uint8().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(int8_t& out) & noexcept { return get_int8().get(out); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::float32_t& out) & noexcept { return get_float32().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::float64_t& out) & noexcept { return get_float64().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<> simdjson_warn_unused simdjson_inline error_code document::get(bool& out) & noexcept { return get_bool().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(value& out) & noexcept { return get_value().get(out); }

 template<> simdjson_deprecated simdjson_inline simdjson_result<raw_json_string> document::get() && noexcept { return get_raw_json_string(); }
 template<> simdjson_deprecated simdjson_inline simdjson_result<std::string_view> document::get() && noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_deprecated simdjson_inline simdjson_result<std::u8string_view> document::get() && noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_deprecated simdjson_inline simdjson_result<double> document::get() && noexcept { return std::forward<document>(*this).get_double(); }
+template<> simdjson_deprecated simdjson_inline simdjson_result<float> document::get() && noexcept { return std::forward<document>(*this).get_float(); }
 template<> simdjson_deprecated simdjson_inline simdjson_result<uint64_t> document::get() && noexcept { return std::forward<document>(*this).get_uint64(); }
 template<> simdjson_deprecated simdjson_inline simdjson_result<int64_t> document::get() && noexcept { return std::forward<document>(*this).get_int64(); }
 template<> simdjson_deprecated simdjson_inline simdjson_result<bool> document::get() && noexcept { return std::forward<document>(*this).get_bool(); }
@@ -125252,6 +159834,22 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<ppc64::ondemand::docume
   if (error()) { return error(); }
   return first.get_int32();
 }
+simdjson_inline simdjson_result<uint16_t> simdjson_result<ppc64::ondemand::document>::get_uint16() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<ppc64::ondemand::document>::get_int16() noexcept {
+  if (error()) { return error(); }
+  return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<ppc64::ondemand::document>::get_uint8() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<ppc64::ondemand::document>::get_int8() noexcept {
+  if (error()) { return error(); }
+  return first.get_int8();
+}
 simdjson_inline simdjson_result<double> simdjson_result<ppc64::ondemand::document>::get_double() noexcept {
   if (error()) { return error(); }
   return first.get_double();
@@ -125260,10 +159858,36 @@ simdjson_inline simdjson_result<double> simdjson_result<ppc64::ondemand::documen
   if (error()) { return error(); }
   return first.get_double_in_string();
 }
+simdjson_inline simdjson_result<float> simdjson_result<ppc64::ondemand::document>::get_float() noexcept {
+  if (error()) { return error(); }
+  return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<ppc64::ondemand::document>::get_float_in_string() noexcept {
+  if (error()) { return error(); }
+  return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<ppc64::ondemand::document>::get_float32() noexcept {
+  if (error()) { return error(); }
+  return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<ppc64::ondemand::document>::get_float64() noexcept {
+  if (error()) { return error(); }
+  return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<std::string_view> simdjson_result<ppc64::ondemand::document>::get_string(bool allow_replacement) noexcept {
   if (error()) { return error(); }
   return first.get_string(allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<ppc64::ondemand::document>::get_u8string(bool allow_replacement) noexcept {
+  if (error()) { return error(); }
+  return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code simdjson_result<ppc64::ondemand::document>::get_string(string_type& receiver, bool allow_replacement) noexcept {
   if (error()) { return error(); }
@@ -125291,22 +159915,46 @@ simdjson_inline simdjson_result<bool> simdjson_result<ppc64::ondemand::document>
 }

 template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<ppc64::ondemand::document>::get() & noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<ppc64::ondemand::document>::get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, ppc64::ondemand::document>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>();
 }
 template<typename T>
-simdjson_deprecated simdjson_inline simdjson_result<T> simdjson_result<ppc64::ondemand::document>::get() && noexcept {
+simdjson_deprecated simdjson_inline simdjson_result<T> simdjson_result<ppc64::ondemand::document>::get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, ppc64::ondemand::document>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return std::forward<ppc64::ondemand::document>(first).get<T>();
 }
 template<typename T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<ppc64::ondemand::document>::get(T &out) & noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<ppc64::ondemand::document>::get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, ppc64::ondemand::document>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>(out);
 }
 template<typename T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<ppc64::ondemand::document>::get(T &out) && noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<ppc64::ondemand::document>::get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, ppc64::ondemand::document>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return std::forward<ppc64::ondemand::document>(first).get<T>(out);
 }
@@ -125375,27 +160023,27 @@ simdjson_inline simdjson_result<ppc64::ondemand::document>::operator ppc64::onde
 }
 simdjson_inline simdjson_result<ppc64::ondemand::document>::operator uint64_t() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_uint64();
 }
 simdjson_inline simdjson_result<ppc64::ondemand::document>::operator int64_t() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_int64();
 }
 simdjson_inline simdjson_result<ppc64::ondemand::document>::operator double() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_double();
 }
 simdjson_inline simdjson_result<ppc64::ondemand::document>::operator std::string_view() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_string();
 }
 simdjson_inline simdjson_result<ppc64::ondemand::document>::operator ppc64::ondemand::raw_json_string() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_raw_json_string();
 }
 simdjson_inline simdjson_result<ppc64::ondemand::document>::operator bool() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_bool();
 }
 simdjson_inline simdjson_result<ppc64::ondemand::document>::operator ppc64::ondemand::value() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
@@ -125485,21 +160133,38 @@ simdjson_inline simdjson_result<uint64_t> document_reference::get_uint64() noexc
 simdjson_inline simdjson_result<uint64_t> document_reference::get_uint64_in_string() noexcept { return doc->get_root_value_iterator().get_root_uint64_in_string(false); }
 simdjson_inline simdjson_result<int64_t> document_reference::get_int64() noexcept { return doc->get_root_value_iterator().get_root_int64(false); }
 simdjson_inline simdjson_result<int64_t> document_reference::get_int64_in_string() noexcept { return doc->get_root_value_iterator().get_root_int64_in_string(false); }
-simdjson_inline simdjson_result<uint32_t> document_reference::get_uint32() noexcept {
-  uint64_t result;
-  SIMDJSON_TRY(doc->get_root_value_iterator().get_root_uint64(false).get(result));
-  if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<uint32_t>(result);
-}
-simdjson_inline simdjson_result<int32_t> document_reference::get_int32() noexcept {
-  int64_t result;
-  SIMDJSON_TRY(doc->get_root_value_iterator().get_root_int64(false).get(result));
-  if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<int32_t>(result);
-}
+simdjson_inline simdjson_result<uint32_t> document_reference::get_uint32() noexcept { return narrow_integer<uint32_t>(get_uint64()); }
+simdjson_inline simdjson_result<int32_t> document_reference::get_int32() noexcept { return narrow_integer<int32_t>(get_int64()); }
+simdjson_inline simdjson_result<uint16_t> document_reference::get_uint16() noexcept { return narrow_integer<uint16_t>(get_uint64()); }
+simdjson_inline simdjson_result<int16_t> document_reference::get_int16() noexcept { return narrow_integer<int16_t>(get_int64()); }
+simdjson_inline simdjson_result<uint8_t> document_reference::get_uint8() noexcept { return narrow_integer<uint8_t>(get_uint64()); }
+simdjson_inline simdjson_result<int8_t> document_reference::get_int8() noexcept { return narrow_integer<int8_t>(get_int64()); }
 simdjson_inline simdjson_result<double> document_reference::get_double() noexcept { return doc->get_root_value_iterator().get_root_double(false); }
-simdjson_inline simdjson_result<double> document_reference::get_double_in_string() noexcept { return doc->get_root_value_iterator().get_root_double(false); }
+simdjson_inline simdjson_result<double> document_reference::get_double_in_string() noexcept { return doc->get_root_value_iterator().get_root_double_in_string(false); }
+simdjson_inline simdjson_result<float> document_reference::get_float() noexcept { return doc->get_root_value_iterator().get_root_float(false); }
+simdjson_inline simdjson_result<float> document_reference::get_float_in_string() noexcept { return doc->get_root_value_iterator().get_root_float_in_string(false); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> document_reference::get_float32() noexcept {
+  float result;
+  SIMDJSON_TRY(get_float().get(result));
+  return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> document_reference::get_float64() noexcept {
+  double result;
+  SIMDJSON_TRY(get_double().get(result));
+  return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<std::string_view> document_reference::get_string(bool allow_replacement) noexcept { return doc->get_root_value_iterator().get_root_string(false, allow_replacement); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> document_reference::get_u8string(bool allow_replacement) noexcept {
+  std::string_view content;
+  SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+  return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code document_reference::get_string(string_type& receiver, bool allow_replacement) noexcept { return doc->get_root_value_iterator().get_root_string(receiver, false, allow_replacement); }
 simdjson_inline simdjson_result<std::string_view> document_reference::get_wobbly_string() noexcept { return doc->get_root_value_iterator().get_root_wobbly_string(false); }
@@ -125511,11 +160176,25 @@ template<> simdjson_inline simdjson_result<array> document_reference::get() & no
 template<> simdjson_inline simdjson_result<object> document_reference::get() & noexcept { return get_object(); }
 template<> simdjson_inline simdjson_result<raw_json_string> document_reference::get() & noexcept { return get_raw_json_string(); }
 template<> simdjson_inline simdjson_result<std::string_view> document_reference::get() & noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> document_reference::get() & noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_inline simdjson_result<double> document_reference::get() & noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> document_reference::get() & noexcept { return get_float(); }
 template<> simdjson_inline simdjson_result<uint64_t> document_reference::get() & noexcept { return get_uint64(); }
 template<> simdjson_inline simdjson_result<int64_t> document_reference::get() & noexcept { return get_int64(); }
 template<> simdjson_inline simdjson_result<uint32_t> document_reference::get() & noexcept { return get_uint32(); }
 template<> simdjson_inline simdjson_result<int32_t> document_reference::get() & noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> document_reference::get() & noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> document_reference::get() & noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> document_reference::get() & noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> document_reference::get() & noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> document_reference::get() & noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> document_reference::get() & noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<> simdjson_inline simdjson_result<bool> document_reference::get() & noexcept { return get_bool(); }
 template<> simdjson_inline simdjson_result<value> document_reference::get() & noexcept { return get_value(); }
 #if SIMDJSON_EXCEPTIONS
@@ -125661,6 +160340,22 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<ppc64::ondemand::docume
   if (error()) { return error(); }
   return first.get_int32();
 }
+simdjson_inline simdjson_result<uint16_t> simdjson_result<ppc64::ondemand::document_reference>::get_uint16() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<ppc64::ondemand::document_reference>::get_int16() noexcept {
+  if (error()) { return error(); }
+  return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<ppc64::ondemand::document_reference>::get_uint8() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<ppc64::ondemand::document_reference>::get_int8() noexcept {
+  if (error()) { return error(); }
+  return first.get_int8();
+}
 simdjson_inline simdjson_result<double> simdjson_result<ppc64::ondemand::document_reference>::get_double() noexcept {
   if (error()) { return error(); }
   return first.get_double();
@@ -125669,10 +160364,36 @@ simdjson_inline simdjson_result<double> simdjson_result<ppc64::ondemand::documen
   if (error()) { return error(); }
   return first.get_double_in_string();
 }
+simdjson_inline simdjson_result<float> simdjson_result<ppc64::ondemand::document_reference>::get_float() noexcept {
+  if (error()) { return error(); }
+  return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<ppc64::ondemand::document_reference>::get_float_in_string() noexcept {
+  if (error()) { return error(); }
+  return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<ppc64::ondemand::document_reference>::get_float32() noexcept {
+  if (error()) { return error(); }
+  return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<ppc64::ondemand::document_reference>::get_float64() noexcept {
+  if (error()) { return error(); }
+  return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<std::string_view> simdjson_result<ppc64::ondemand::document_reference>::get_string(bool allow_replacement) noexcept {
   if (error()) { return error(); }
   return first.get_string(allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<ppc64::ondemand::document_reference>::get_u8string(bool allow_replacement) noexcept {
+  if (error()) { return error(); }
+  return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code simdjson_result<ppc64::ondemand::document_reference>::get_string(string_type& receiver, bool allow_replacement) noexcept {
   if (error()) { return error(); }
@@ -125699,22 +160420,46 @@ simdjson_inline simdjson_result<bool> simdjson_result<ppc64::ondemand::document_
   return first.is_null();
 }
 template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<ppc64::ondemand::document_reference>::get() & noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<ppc64::ondemand::document_reference>::get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, ppc64::ondemand::document_reference>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>();
 }
 template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<ppc64::ondemand::document_reference>::get() && noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<ppc64::ondemand::document_reference>::get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, ppc64::ondemand::document_reference>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return std::forward<ppc64::ondemand::document_reference>(first).get<T>();
 }
 template <class T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<ppc64::ondemand::document_reference>::get(T &out) & noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<ppc64::ondemand::document_reference>::get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, ppc64::ondemand::document_reference>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>(out);
 }
 template <class T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<ppc64::ondemand::document_reference>::get(T &out) && noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<ppc64::ondemand::document_reference>::get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, ppc64::ondemand::document_reference>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return std::forward<ppc64::ondemand::document_reference>(first).get<T>(out);
 }
@@ -125776,27 +160521,27 @@ simdjson_inline simdjson_result<ppc64::ondemand::document_reference>::operator p
 }
 simdjson_inline simdjson_result<ppc64::ondemand::document_reference>::operator uint64_t() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_uint64();
 }
 simdjson_inline simdjson_result<ppc64::ondemand::document_reference>::operator int64_t() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_int64();
 }
 simdjson_inline simdjson_result<ppc64::ondemand::document_reference>::operator double() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_double();
 }
 simdjson_inline simdjson_result<ppc64::ondemand::document_reference>::operator std::string_view() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_string();
 }
 simdjson_inline simdjson_result<ppc64::ondemand::document_reference>::operator ppc64::ondemand::raw_json_string() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_raw_json_string();
 }
 simdjson_inline simdjson_result<ppc64::ondemand::document_reference>::operator bool() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_bool();
 }
 simdjson_inline simdjson_result<ppc64::ondemand::document_reference>::operator ppc64::ondemand::value() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
@@ -125862,6 +160607,7 @@ simdjson_warn_unused simdjson_inline error_code simdjson_result<ppc64::ondemand:
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */

 #include <algorithm>
+#include <cstring>
 #include <stdexcept>

 namespace simdjson {
@@ -125948,23 +160694,20 @@ simdjson_inline document_stream::document_stream(
   const uint8_t *_buf,
   size_t _len,
   size_t _batch_size,
-  bool _allow_comma_separated
+  bool _allow_comma_separated,
+  stream_format _format
 ) noexcept
   : parser{&_parser},
     buf{_buf},
     len{_len},
     batch_size{_batch_size <= MINIMAL_BATCH_SIZE ? MINIMAL_BATCH_SIZE : _batch_size},
     allow_comma_separated{_allow_comma_separated},
+    format{_format},
     error{SUCCESS}
     #ifdef SIMDJSON_THREADS_ENABLED
     , use_thread(_parser.threaded) // we need to make a copy because _parser.threaded can change
     #endif
 {
-#ifdef SIMDJSON_THREADS_ENABLED
-  if(worker.get() == nullptr) {
-    error = MEMALLOC;
-  }
-#endif
 }

 simdjson_inline document_stream::document_stream() noexcept
@@ -125973,6 +160716,7 @@ simdjson_inline document_stream::document_stream() noexcept
     len{0},
     batch_size{0},
     allow_comma_separated{false},
+    format{stream_format::whitespace_delimited},
     error{UNINITIALIZED}
     #ifdef SIMDJSON_THREADS_ENABLED
     , use_thread(false)
@@ -125992,6 +160736,9 @@ inline size_t document_stream::size_in_bytes() const noexcept {
 }

 inline size_t document_stream::truncated_bytes() const noexcept {
+  // Stage 1 returns EMPTY on zero-length input before it writes the index
+  // sentinels read below, so they would still hold a previous stream's values.
+  if (len == 0) { return 0; }
   if(error == CAPACITY) { return len - batch_start; }
   return parser->implementation->structural_indexes[parser->implementation->n_structural_indexes] - parser->implementation->structural_indexes[parser->implementation->n_structural_indexes + 1];
 }
@@ -126072,13 +160819,20 @@ inline void document_stream::start() noexcept {
     error = run_stage1(*parser, batch_start);
   }
   if (error) { return; }
-  doc_index = batch_start;
+  // For json_sequence mode, structural_indexes[0] points to the actual JSON value
+  // after the RS delimiter and any following whitespace. For regular mode, it is
+  // the offset from batch_start to the first document in the batch.
+  doc_index = batch_start + parser->implementation->structural_indexes[0];
   doc = document(json_iterator(&buf[batch_start], parser));
   doc.iter._streaming = true;

   #ifdef SIMDJSON_THREADS_ENABLED
   if (use_thread && next_batch_start() < len) {
     // Kick off the first thread on next batch if needed
+    if (worker.get() == nullptr) {
+      worker.reset(new(std::nothrow) stage1_worker());
+      if (worker.get() == nullptr) { error = MEMALLOC; return; }
+    }
     error = stage1_thread_parser.allocate(batch_size);
     if (error) { return; }
     worker->start_thread();
@@ -126153,12 +160907,69 @@ inline void document_stream::next() noexcept {
        */

       if (error) { continue; } // If the error was EMPTY, we may want to load another batch.
-      doc_index = batch_start;
+      doc_index = batch_start + parser->implementation->structural_indexes[0];
     }
   }
 }

+simdjson_inline uint8_t document_stream::document_delimiter() const noexcept {
+  switch (format) {
+    case stream_format::newline_delimited: return '\n';
+    case stream_format::json_sequence: return 0x1E;
+    default: return 0;
+  }
+}
+
+simdjson_inline bool document_stream::skip_to_delimiter(uint8_t delimiter) noexcept {
+  const uint8_t *const base = &buf[batch_start];
+  const token_position pos = doc.iter.position();
+  const token_position end = doc.iter.end_position();
+  if (pos >= end) { return false; }
+  const size_t here = size_t(doc.iter.token.peek(pos) - base);
+  const size_t batch_len =
+      (len - batch_start < batch_size) ? len - batch_start : batch_size;
+  if (here >= batch_len) { return false; }
+  const uint8_t *const found = static_cast<const uint8_t *>(
+      std::memchr(base + here, delimiter, batch_len - here));
+  if (found == nullptr) { return false; }
+
+  const uint32_t boundary = uint32_t(found - base);
+  // The answer is near `pos`: the delimiter ends the current document, while
+  // `end` spans the whole batch. Gallop first so the cost follows the distance
+  // rather than the size of the batch.
+  token_position lo = pos;
+  size_t hop = 1;
+  while (lo + hop < end && lo[hop] < boundary) { lo += hop; hop <<= 1; }
+  token_position hi = (lo + hop < end) ? lo + hop : end;
+  while (lo < hi) {
+    const token_position mid = lo + ((hi - lo) >> 1);
+    if (*mid < boundary) { lo = mid + 1; } else { hi = mid; }
+  }
+  doc.iter.token.set_position(lo);
+  return true;
+}
+
 inline void document_stream::next_document() noexcept {
+  // A delimiter that cannot occur inside a document tells us where the current
+  // one ends, so we can jump there instead of walking every structural. Only
+  // valid while the iterator is still inside the document: a consumed document
+  // already sits on the next one's first token, and skip_child() returns at
+  // once for it.
+  //
+  // The jump does not structure-validate the unread remainder of the current
+  // document: under newline_delimited / json_sequence the next delimiter is
+  // assumed to be the true document boundary. Callers that leave depth() > 0
+  // while violating that contract (e.g. pretty multi-line JSON under
+  // newline_delimited) can mis-align following documents; use
+  // whitespace_delimited if unsure.
+  const uint8_t delimiter = document_delimiter();
+  if (delimiter != 0 && !error && doc.iter.depth() > 0 &&
+      skip_to_delimiter(delimiter)) {
+    doc.iter._depth = 1;
+    doc.iter._string_buf_loc = parser->string_buf.get();
+    doc.iter._root = doc.iter.position();
+    return;
+  }
   // Go to next place where depth=0 (document depth)
   error = doc.iter.skip_child(0);
   if (error) { return; }
@@ -126182,10 +160993,35 @@ inline error_code document_stream::run_stage1(ondemand::parser &p, size_t _batch
   // This code only updates the structural index in the parser, it does not update any json_iterator
   // instance.
   size_t remaining = len - _batch_start;
+  stage1_mode mode;
   if (remaining <= batch_size) {
-    return p.implementation->stage1(&buf[_batch_start], remaining, stage1_mode::streaming_final);
+    // Final batch
+    switch (format) {
+      case stream_format::json_sequence:
+        mode = stage1_mode::json_sequence_final;
+        break;
+      case stream_format::comma_delimited:
+        mode = stage1_mode::comma_delimited_final;
+        break;
+      default:
+        mode = stage1_mode::streaming_final;
+        break;
+    }
+    return p.implementation->stage1(&buf[_batch_start], remaining, mode);
   } else {
-    return p.implementation->stage1(&buf[_batch_start], batch_size, stage1_mode::streaming_partial);
+    // Partial batch
+    switch (format) {
+      case stream_format::json_sequence:
+        mode = stage1_mode::json_sequence_partial;
+        break;
+      case stream_format::comma_delimited:
+        mode = stage1_mode::comma_delimited_partial;
+        break;
+      default:
+        mode = stage1_mode::streaming_partial;
+        break;
+    }
+    return p.implementation->stage1(&buf[_batch_start], batch_size, mode);
   }
 }

@@ -126194,11 +161030,19 @@ simdjson_inline size_t document_stream::iterator::current_index() const noexcept
 }

 simdjson_inline std::string_view document_stream::iterator::source() const noexcept {
-  auto depth = stream->doc.iter.depth();
+  // On error (e.g., CAPACITY), there is no document to walk: return the rest of
+  // the input, as the DOM document_stream does.
+  if (stream->error) {
+    return std::string_view(reinterpret_cast<const char*>(stream->buf) + current_index(), stream->len - current_index());
+  }
+  // Always walk from the root of the document, whatever the current position
+  // of the document iterator: the user may have already consumed part of the
+  // document, so the iterator's current depth must not be used here.
+  depth_t depth = 1;
   auto cur_struct_index = stream->doc.iter._root - stream->parser->implementation->structural_indexes.get();

-  // If at root, process the first token to determine if scalar value
-  if (stream->doc.iter.at_root()) {
+  // Process the first token to determine if scalar value
+  {
     switch (stream->buf[stream->batch_start + stream->parser->implementation->structural_indexes[cur_struct_index]]) {
       case '{': case '[':   // Depth=1 already at start of document
         break;
@@ -126206,14 +161050,47 @@ simdjson_inline std::string_view document_stream::iterator::source() const noexc
         depth--;
         break;
       default:    // Scalar value document
-        // TODO: We could remove trailing whitespaces
         // This returns a string spanning from start of value to the beginning of the next document (excluded)
         {
           auto next_index = stream->batch_start + stream->parser->implementation->structural_indexes[++cur_struct_index];
           // normally the length would be next_index - current_index() - 1, except for the last document
           size_t svlen = next_index - current_index();
           const char *start = reinterpret_cast<const char*>(stream->buf) + current_index();
-          while(svlen > 1 && (std::isspace(start[svlen-1]) || start[svlen-1] == '\0')) {
+          // When the scalar is followed by a truncated document, the structural
+          // indexes of that document were dropped and next_index is the end of
+          // the input, so we bound the scalar by scanning the token itself.
+          size_t token_len = 0;
+          if (*start == '"') {
+            token_len = 1;
+            while (token_len < svlen) {
+              char c = start[token_len++];
+              if (c == '\\') {
+                token_len++;
+              } else if (c == '"') {
+                break;
+              }
+            }
+          } else {
+            while (token_len < svlen) {
+              char c = start[token_len];
+              if (std::isspace(static_cast<unsigned char>(c)) || c == ',' || c == '{' || c == '[' || c == '\0' || static_cast<uint8_t>(c) == 0x1E) {
+                break;
+              }
+              token_len++;
+            }
+          }
+          if (token_len > 0 && token_len < svlen) {
+            svlen = token_len;
+          }
+          // Trim trailing whitespace, NUL, and RS (0x1E). In RFC 7464
+          // json_sequence mode the scanner classifies RS as a scalar
+          // character, so an RS-prefixed scalar document (number / true /
+          // false / null / string) has no closing structural index and the
+          // slice runs all the way up to the next document's RS. RS cannot
+          // legally appear in a JSON value at the source level (control
+          // characters in strings must be escaped as \u001E), so stripping
+          // it is safe in every stream_format.
+          while(svlen > 1 && (std::isspace(static_cast<unsigned char>(start[svlen-1])) || start[svlen-1] == '\0' || static_cast<uint8_t>(start[svlen-1]) == 0x1E || (stream->format == stream_format::comma_delimited && start[svlen-1] == ','))) {
             svlen--;
           }
           return std::string_view(start, svlen);
@@ -126338,11 +161215,19 @@ simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> field::un
   return answer;
 }

+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_warn_unused simdjson_result<std::u8string_view> field::unescaped_u8key(bool allow_replacement) noexcept {
+  std::string_view key;
+  SIMDJSON_TRY( unescaped_key(allow_replacement).get(key) );
+  return internal::as_u8string_view(key);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 template <typename string_type>
 simdjson_inline simdjson_warn_unused error_code field::unescaped_key(string_type& receiver, bool allow_replacement) noexcept {
   std::string_view key;
   SIMDJSON_TRY( unescaped_key(allow_replacement).get(key) );
-  receiver = key;
+  internal::assign_utf8(receiver, key);
   return SUCCESS;
 }

@@ -126364,6 +161249,12 @@ simdjson_inline std::string_view field::escaped_key() const noexcept {
   return std::string_view(reinterpret_cast<const char*>(first.buf), end_quote - first.buf);
 }

+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline std::u8string_view field::escaped_u8key() const noexcept {
+  return internal::as_u8string_view(escaped_key());
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 simdjson_inline value &field::value() & noexcept {
   return second;
 }
@@ -126408,11 +161299,25 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<ppc64::ondeman
   return first.escaped_key();
 }

+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<ppc64::ondemand::field>::escaped_u8key() noexcept {
+  if (error()) { return error(); }
+  return first.escaped_u8key();
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 simdjson_inline simdjson_result<std::string_view> simdjson_result<ppc64::ondemand::field>::unescaped_key(bool allow_replacement) noexcept {
   if (error()) { return error(); }
   return first.unescaped_key(allow_replacement);
 }

+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<ppc64::ondemand::field>::unescaped_u8key(bool allow_replacement) noexcept {
+  if (error()) { return error(); }
+  return first.unescaped_u8key(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 template<typename string_type>
 simdjson_warn_unused simdjson_inline error_code simdjson_result<ppc64::ondemand::field>::unescaped_key(string_type &receiver, bool allow_replacement) noexcept {
   if (error()) { return error(); }
@@ -126456,6 +161361,9 @@ simdjson_inline json_iterator::json_iterator(json_iterator &&other) noexcept
     _depth{other._depth},
     _root{other._root},
     _streaming{other._streaming}
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+    , _allow_incomplete_json{other._allow_incomplete_json}
+#endif
 {
   other.parser = nullptr;
 }
@@ -126467,6 +161375,9 @@ simdjson_inline json_iterator &json_iterator::operator=(json_iterator &&other) n
   _depth = other._depth;
   _root = other._root;
   _streaming = other._streaming;
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  _allow_incomplete_json = other._allow_incomplete_json;
+#endif
   other.parser = nullptr;
   return *this;
 }
@@ -126493,7 +161404,8 @@ simdjson_inline json_iterator::json_iterator(const uint8_t *buf, ondemand::parse
       _string_buf_loc{parser->string_buf.get()},
       _depth{1},
       _root{parser->implementation->structural_indexes.get()},
-      _streaming{streaming}
+      _streaming{streaming},
+      _allow_incomplete_json{true}

 {
   logger::log_headers();
@@ -126565,7 +161477,8 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::skip_child(depth_
 #endif // SIMDJSON_CHECK_EOF
       break;
     case '"':
-      if(*peek() == ':') {
+      // At the end, peek() would read the sentinel, which points into the padding.
+      if(!at_end() && *peek() == ':') {
         // We are at a key!!!
         // This might happen if you just started an object and you skip it immediately.
         // Performance note: it would be nice to get rid of this check as it is somewhat
@@ -126608,7 +161521,7 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::skip_child(depth_
     }
   }

-  return report_error(TAPE_ERROR, "not enough close braces");
+  return report_error(INCOMPLETE_ARRAY_OR_OBJECT, "not enough close braces");
 }

 SIMDJSON_POP_DISABLE_WARNINGS
@@ -126625,6 +161538,17 @@ simdjson_inline bool json_iterator::streaming() const noexcept {
   return _streaming;
 }

+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+simdjson_inline bool json_iterator::allow_incomplete_json() const noexcept {
+  return _allow_incomplete_json;
+}
+
+simdjson_inline size_t json_iterator::remaining_input_length(const uint8_t *json) const noexcept {
+  const uint8_t *end = token.buf + parser->_document_len;
+  return json < end ? size_t(end - json) : 0;
+}
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+
 simdjson_inline token_position json_iterator::root_position() const noexcept {
   return _root;
 }
@@ -126907,7 +161831,7 @@ inline std::ostream& operator<<(std::ostream& out, json_type type) noexcept {
         case json_type::string: out << "string"; break;
         case json_type::boolean: out << "boolean"; break;
         case json_type::null: out << "null"; break;
-        default: SIMDJSON_UNREACHABLE();
+        case json_type::unknown: out << "unknown"; break;
     }
     return out;
 }
@@ -127246,6 +162170,10 @@ inline void log_line(const json_iterator &iter, token_position index, depth_t de
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_iterator.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value-inl.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/jsonpathutil.h" */
+/* amalgamation skipped (editor-only): #include <utility> */
+/* amalgamation skipped (editor-only): #if SIMDJSON_SUPPORTS_CONCEPTS */
+/* amalgamation skipped (editor-only): #include <tuple> // std::forward_as_tuple/get for the variadic for_each adapter */
+/* amalgamation skipped (editor-only): #endif */
 /* amalgamation skipped (editor-only): #if SIMDJSON_STATIC_REFLECTION */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_string_builder.h"  // for constevalutil::fixed_string */
 /* amalgamation skipped (editor-only): #include <meta> */
@@ -127275,12 +162203,21 @@ simdjson_inline simdjson_result<value> object::find_field_unordered(const std::s
   return value(iter.child());
 }
 simdjson_inline simdjson_result<value> object::operator[](const std::string_view key) & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   return find_field_unordered(key);
 }
 simdjson_inline simdjson_result<value> object::operator[](const std::string_view key) && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   return std::forward<object>(*this).find_field_unordered(key);
 }
 simdjson_inline simdjson_result<value> object::find_field(const std::string_view key) & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   bool has_value;
   SIMDJSON_TRY( iter.find_field_raw(key).get(has_value) );
   if (!has_value) {
@@ -127290,6 +162227,9 @@ simdjson_inline simdjson_result<value> object::find_field(const std::string_view
   return value(iter.child());
 }
 simdjson_inline simdjson_result<value> object::find_field(const std::string_view key) && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   bool has_value;
   SIMDJSON_TRY( iter.find_field_raw(key).get(has_value) );
   if (!has_value) {
@@ -127299,6 +162239,150 @@ simdjson_inline simdjson_result<value> object::find_field(const std::string_view
   return value(iter.child());
 }

+#if SIMDJSON_SUPPORTS_CONCEPTS
+template <typename Selector, typename Func>
+  requires key_selector_type<Selector> &&
+           std::is_invocable_v<Func&, std::size_t, value>
+simdjson_flatten simdjson_inline for_each_result object::for_each(Func&& on_match)
+    noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, value>) {
+  // Single pass driven directly by the value_iterator, mirroring
+  // find_field_unordered_raw + value(iter.child()). Compared to walking via
+  // object_iterator/field, this avoids constructing a simdjson_result<field> and
+  // a field (key + value) for every field -- and the development-check bookkeeping
+  // in object_iterator -- building a value only for the fields that actually match.
+  // We operate on a copy of the iterator, as object::begin() would.
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  // Mirror object::begin(): for_each must start at the beginning of the object,
+  // not from some position left behind by a prior find_field on the same object.
+  if (!iter.is_at_iterator_start()) { return {OUT_OF_ORDER_ITERATION, 0}; }
+#endif
+  value_iterator it = iter;
+  std::size_t matched = 0;
+  // Track which selector indices have already matched, as a compile-time bitset
+  // (one 64-bit word per 64 keys): the callback fires once per key, on its first
+  // occurrence, and we stop as soon as every key has matched.
+  constexpr std::size_t seen_words = (Selector::size() + 63) / 64;
+  std::array<std::uint64_t, seen_words> seen{};
+  while (it.is_open()) {
+    raw_json_string key;
+    error_code error;
+    std::size_t idx;
+    if constexpr (Selector::window.ok) {
+      // A window selector confirms a key from its raw bytes alone (the closing
+      // quote bounds it), so we take the length-free path: field_key (no backward
+      // scan, mirroring the ordered find_field) feeding match_raw(raw_json_string).
+      if ((error = it.field_key().get(key))) { it.abandon(); return {error, matched}; }
+      if ((error = it.field_value())) { it.abandon(); return {error, matched}; }
+      idx = Selector::match_raw(key);
+    } else {
+      // Otherwise derive the key length from the structural index (the following
+      // ':' token) to feed the length-aware hash, avoiding a forward SIMD scan.
+      std::size_t key_len;
+      if ((error = it.field_key_with_length(key, key_len))) { it.abandon(); return {error, matched}; }
+      if ((error = it.field_value())) { it.abandon(); return {error, matched}; }
+      idx = Selector::match_raw(key.raw(), key_len);
+    }
+    if (idx < Selector::size()) {
+      const std::uint64_t seen_bit = std::uint64_t{1} << (idx & 63);
+      std::uint64_t &seen_word = seen[idx >> 6];
+      if (!(seen_word & seen_bit)) {
+        seen_word |= seen_bit;
+        value matched_value(it.child());
+        // The callback may return void or anything convertible to error_code
+        // (error_code itself, or a for_each_result from a nested for_each). When
+        // it yields an error_code, we stop at the first non-SUCCESS result and
+        // propagate it so the caller can surface value-parse errors (for example,
+        // a type mismatch on a matched field). A void-returning callback is
+        // responsible for handling its own errors.
+        if constexpr (std::is_convertible_v<decltype(on_match(idx, matched_value)), error_code>) {
+          // Unlike the internal-error paths above, a callback error does not
+          // abandon the iterator: we leave it recoverable so the caller can keep
+          // using the object (or its parent) after handling the error.
+          if ((error = on_match(idx, matched_value))) { return {error, matched}; }
+        } else {
+          on_match(idx, matched_value);
+        }
+        if (++matched >= Selector::size()) { break; }
+      }
+    }
+    // Mirror object_iterator::operator++'s safety rail: if the callback consumed
+    // the value and left the iterator closed or in error (e.g. a void callback
+    // that swallowed a fatal sub-iteration error), stop here rather than calling
+    // skip_child on a closed iterator.
+    if (!it.is_open()) { break; }
+    // Skip the value (a no-op if the callback consumed it) and step to the next
+    // field; has_next_field() ends the container on '}', which closes the loop.
+    if ((error = it.skip_child())) { it.abandon(); return {error, matched}; }
+    if ((error = it.has_next_field().error())) { return {error, matched}; }
+  }
+  return {SUCCESS, matched};
+}
+
+namespace key_selector_for_each_detail {
+
+// Dispatch a matched value to the I-th handler in the tuple (0-based). A handler
+// is either an invocable taking the value (void- or error_code-returning) or a
+// deserialization target, in which case we assign via value::get. We always
+// return error_code so the core (index, value) for_each can treat the adapter
+// uniformly.
+template <typename Tuple, std::size_t... Is>
+simdjson_really_inline error_code dispatch_value(
+    std::size_t idx, Tuple& handlers, value v, std::index_sequence<Is...>) {
+  error_code err = SUCCESS;
+  auto try_one = [&](auto Ic) {
+    constexpr std::size_t I = decltype(Ic)::value;
+    if (idx == I) {
+      auto&& h = std::get<I>(handlers);
+      using H = std::remove_reference_t<decltype(h)>;
+      if constexpr (std::is_invocable_v<H&, value>) {
+        // A handler returning void runs for its side effects; one returning
+        // anything convertible to error_code (error_code, or a for_each_result
+        // from a nested for_each) has its error captured and propagated.
+        if constexpr (std::is_convertible_v<decltype(h(v)), error_code>) {
+          err = h(v);
+        } else {
+          h(v);
+        }
+      } else {
+        // Direct deserialization target: assign the matched value into it.
+        err = v.get(h);
+      }
+    }
+  };
+  (try_one(std::integral_constant<std::size_t, Is>{}), ...);
+  return err;
+}
+
+} // namespace key_selector_for_each_detail
+
+template <typename Selector, typename... Handlers>
+  requires key_selector_type<Selector> &&
+           (sizeof...(Handlers) == Selector::size()) &&
+           (key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_flatten simdjson_inline for_each_result object::for_each(Handlers&&... on_match)
+    noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+  auto handlers = std::forward_as_tuple(std::forward<Handlers>(on_match)...);
+  // Reuse the single (index, value) implementation via a tiny adapter.
+  // The adapter is called once per *matched* key (very few); the hot path
+  // (iteration + match_raw + seen bitset) stays exactly the same.
+  return this->template for_each<Selector>(
+      [&](std::size_t i, value v) -> error_code {
+        return key_selector_for_each_detail::dispatch_value(
+            i, handlers, v, std::make_index_sequence<Selector::size()>{});
+      });
+}
+
+template <constevalutil::fixed_string... Keys, typename... Handlers>
+  requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+           (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+           (key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline for_each_result object::for_each(Handlers&&... on_match)
+    noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+  using Selector = key_selector<Keys...>;
+  return this->template for_each<Selector>(std::forward<Handlers>(on_match)...);
+}
+#endif
+
 simdjson_inline simdjson_result<object> object::start(value_iterator &iter) noexcept {
   SIMDJSON_TRY( iter.start_object().error() );
   return object(iter);
@@ -127334,6 +162418,9 @@ simdjson_warn_unused simdjson_inline error_code object::consume() noexcept {
 }

 simdjson_inline simdjson_result<std::string_view> object::raw_json() noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   const uint8_t * starting_point{iter.peek_start()};
   auto error = consume();
   if(error) { return error; }
@@ -127355,9 +162442,17 @@ simdjson_inline object::object(const value_iterator &_iter) noexcept
 {
 }

-simdjson_inline simdjson_result<object_iterator> object::begin() noexcept {
+simdjson_inline simdjson_result<object_iterator> object::begin() & noexcept {
 #if SIMDJSON_DEVELOPMENT_CHECKS
-  if (!iter.is_at_iterator_start()) { return OUT_OF_ORDER_ITERATION; }
+  if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
+  return object_iterator(iter, this);
+#endif
+  return object_iterator(iter);
+}
+simdjson_inline simdjson_result<object_iterator> object::begin() && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  // The object is a temporary that the iterator may outlive: do not lock it.
+  if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
 #endif
   return object_iterator(iter);
 }
@@ -127366,7 +162461,9 @@ simdjson_inline simdjson_result<object_iterator> object::end() noexcept {
 }

 inline simdjson_result<value> object::at_pointer(std::string_view json_pointer) noexcept {
-  if (json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
+  // An empty pointer has no json_pointer[0]: with a default-constructed
+  // std::string_view, reading it dereferences a null pointer.
+  if (json_pointer.empty() || json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
   json_pointer = json_pointer.substr(1);
   size_t slash = json_pointer.find('/');
   std::string_view key = json_pointer.substr(0, slash);
@@ -127468,6 +162565,9 @@ simdjson_inline simdjson_result<size_t> object::count_fields() & noexcept {
 }

 simdjson_inline simdjson_result<bool> object::is_empty() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   bool is_not_empty;
   auto error = iter.reset_object().get(is_not_empty);
   if(error) { return error; }
@@ -127475,9 +162575,41 @@ simdjson_inline simdjson_result<bool> object::is_empty() & noexcept {
 }

 simdjson_inline simdjson_result<bool> object::reset() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   return iter.reset_object();
 }

+simdjson_inline object_position object::get_current_position() const noexcept {
+  return object_position{iter.position(), iter.json_iter().depth()};
+}
+
+simdjson_inline error_code object::revert_position(object_position position) noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
+  // json_iterator::reenter_child() requires the live depth to be exactly
+  // one level shallower than the target (matching how every other depth
+  // transition in this iterator works), and under SIMDJSON_DEVELOPMENT_CHECKS
+  // additionally validates against the parser's per-depth container-start
+  // bookkeeping. Neither applies here: depending on what was captured and
+  // what has happened since (a scalar field fully consumed, a compound
+  // value left open, a find_field() miss that scanned past everything),
+  // the live depth when reverting can be any number of levels away from
+  // the captured one, and the captured depth is not necessarily a
+  // container's own start. reenter_at() moves directly, matching how
+  // reset_object() itself repositions without going through reenter_child().
+  iter.reenter_at(position.position, position.depth);
+  return SUCCESS;
+}
+
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline void object::set_locked(bool _locked) noexcept {
+  locked = _locked;
+}
+#endif
+
 #if SIMDJSON_SUPPORTS_CONCEPTS && SIMDJSON_STATIC_REFLECTION

 template<constevalutil::fixed_string... FieldNames, typename T>
@@ -127535,10 +162667,14 @@ simdjson_inline simdjson_result<ppc64::ondemand::object>::simdjson_result(ppc64:
 simdjson_inline simdjson_result<ppc64::ondemand::object>::simdjson_result(error_code error) noexcept
     : implementation_simdjson_result_base<ppc64::ondemand::object>(error) {}

-simdjson_inline simdjson_result<ppc64::ondemand::object_iterator> simdjson_result<ppc64::ondemand::object>::begin() noexcept {
+simdjson_inline simdjson_result<ppc64::ondemand::object_iterator> simdjson_result<ppc64::ondemand::object>::begin() & noexcept {
   if (error()) { return error(); }
   return first.begin();
 }
+simdjson_inline simdjson_result<ppc64::ondemand::object_iterator> simdjson_result<ppc64::ondemand::object>::begin() && noexcept {
+  if (error()) { return error(); }
+  return std::move(first).begin();
+}
 simdjson_inline simdjson_result<ppc64::ondemand::object_iterator> simdjson_result<ppc64::ondemand::object>::end() noexcept {
   if (error()) { return error(); }
   return first.end();
@@ -127592,11 +162728,55 @@ simdjson_inline error_code simdjson_result<ppc64::ondemand::object>::for_each_at
   return first.for_each_at_path_with_wildcard(json_path, std::forward<Func>(callback));
 }

+#if SIMDJSON_SUPPORTS_CONCEPTS
+template <typename Selector, typename Func>
+  requires ppc64::ondemand::key_selector_type<Selector> &&
+           std::is_invocable_v<Func&, std::size_t, ppc64::ondemand::value>
+simdjson_inline ppc64::ondemand::for_each_result
+simdjson_result<ppc64::ondemand::object>::for_each(Func&& on_match)
+    noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, ppc64::ondemand::value>) {
+  if (error()) { return {error(), 0}; }
+  return first.template for_each<Selector>(std::forward<Func>(on_match));
+}
+
+template <typename Selector, typename... Handlers>
+  requires ppc64::ondemand::key_selector_type<Selector> &&
+           (sizeof...(Handlers) == Selector::size()) &&
+           (ppc64::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline ppc64::ondemand::for_each_result
+simdjson_result<ppc64::ondemand::object>::for_each(Handlers&&... on_match)
+    noexcept(ppc64::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+  if (error()) { return {error(), 0}; }
+  return first.template for_each<Selector>(std::forward<Handlers>(on_match)...);
+}
+
+template <constevalutil::fixed_string... Keys, typename... Handlers>
+  requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+           (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+           (ppc64::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline ppc64::ondemand::for_each_result
+simdjson_result<ppc64::ondemand::object>::for_each(Handlers&&... on_match)
+    noexcept(ppc64::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+  if (error()) { return {error(), 0}; }
+  return first.template for_each<Keys...>(std::forward<Handlers>(on_match)...);
+}
+#endif
+
 inline simdjson_result<bool> simdjson_result<ppc64::ondemand::object>::reset() noexcept {
   if (error()) { return error(); }
   return first.reset();
 }

+inline simdjson_result<ppc64::ondemand::object_position> simdjson_result<ppc64::ondemand::object>::get_current_position() noexcept {
+  if (error()) { return error(); }
+  return first.get_current_position();
+}
+
+inline error_code simdjson_result<ppc64::ondemand::object>::revert_position(ppc64::ondemand::object_position position) noexcept {
+  if (error()) { return error(); }
+  return first.revert_position(position);
+}
+
 inline simdjson_result<bool> simdjson_result<ppc64::ondemand::object>::is_empty() noexcept {
   if (error()) { return error(); }
   return first.is_empty();
@@ -127640,6 +162820,61 @@ simdjson_inline object_iterator::object_iterator(const value_iterator &_iter) no
   : iter{_iter}
 {}

+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline object_iterator::object_iterator(const value_iterator &_iter, object* _parent) noexcept
+  : parent{_parent}, iter{_iter}
+{
+  if (parent) parent->set_locked(true);
+}
+#endif
+
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline object_iterator::~object_iterator() noexcept
+{
+  if (parent) parent->set_locked(false);
+}
+
+simdjson_inline object_iterator::object_iterator(object_iterator&& other) noexcept
+  : has_been_referenced{other.has_been_referenced},
+    parent{other.parent},
+    iter{std::move(other.iter)}
+{
+  other.parent = nullptr;
+}
+
+simdjson_inline object_iterator& object_iterator::operator=(object_iterator&& other) noexcept {
+  if (this != &other)
+  {
+    if (parent)
+      parent->set_locked(false);
+    has_been_referenced = other.has_been_referenced;
+    parent = other.parent;
+    iter = std::move(other.iter);
+
+    other.parent = nullptr;
+  }
+  return *this;
+}
+
+simdjson_inline object_iterator::object_iterator(const object_iterator& other) noexcept
+  : has_been_referenced{other.has_been_referenced},
+    parent{nullptr},
+    iter{other.iter}
+{}
+
+simdjson_inline object_iterator& object_iterator::operator=(const object_iterator& other) noexcept {
+  if (this != &other)
+  {
+    if (parent)
+      parent->set_locked(false);
+    has_been_referenced = other.has_been_referenced;
+    parent = nullptr;
+    iter = other.iter;
+  }
+  return *this;
+}
+#endif
+
 simdjson_inline simdjson_result<field> object_iterator::operator*() noexcept {
 #if SIMDJSON_DEVELOPMENT_CHECKS
   // We must call * once per iteration.
@@ -127767,6 +163002,147 @@ simdjson_inline simdjson_result<ppc64::ondemand::object_iterator> &simdjson_resu

 #endif // SIMDJSON_GENERIC_ONDEMAND_OBJECT_ITERATOR_INL_H
 /* end file simdjson/generic/ondemand/object_iterator-inl.h for ppc64 */
+/* including simdjson/generic/ondemand/ranges-inl.h for ppc64: #include "simdjson/generic/ondemand/ranges-inl.h" */
+/* begin file simdjson/generic/ondemand/ranges-inl.h for ppc64 */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/ranges.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array_iterator-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object_iterator-inl.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#if SIMDJSON_SUPPORTS_RANGES
+
+namespace simdjson {
+namespace ppc64 {
+namespace ondemand {
+
+//
+// array_range_iterator
+//
+
+simdjson_inline array_range_iterator::array_range_iterator(array_iterator iter) noexcept
+  : iter_{iter} {}
+
+simdjson_inline simdjson_result<value> array_range_iterator::operator*() const noexcept {
+  return *iter_;
+}
+
+simdjson_inline array_range_iterator& array_range_iterator::operator++() noexcept {
+  ++iter_;
+  return *this;
+}
+
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+simdjson_inline void array_range_iterator::operator++(int) noexcept {
+  ++*this;
+}
+SIMDJSON_POP_DISABLE_WARNINGS
+
+//
+// array_range
+//
+
+simdjson_inline array_range::array_range(array& arr) noexcept {
+  auto b = arr.begin();
+  if (b.error()) { error_ = b.error(); return; }
+  begin_ = b.value_unsafe();
+  end_ = arr.end().value_unsafe();
+}
+
+simdjson_inline array_range_iterator array_range::begin() noexcept {
+  return array_range_iterator(begin_);
+}
+
+simdjson_inline array_range_iterator array_range::end() noexcept {
+  return array_range_iterator(end_);
+}
+
+//
+// object_range_iterator
+//
+
+simdjson_inline object_range_iterator::object_range_iterator(object_iterator iter) noexcept
+  : iter_{iter} {}
+
+simdjson_inline simdjson_result<field> object_range_iterator::operator*() const noexcept {
+  return *iter_;
+}
+
+simdjson_inline object_range_iterator& object_range_iterator::operator++() noexcept {
+  ++iter_;
+  return *this;
+}
+
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+simdjson_inline void object_range_iterator::operator++(int) noexcept {
+  ++*this;
+}
+SIMDJSON_POP_DISABLE_WARNINGS
+
+
+//
+// object_range
+//
+
+simdjson_inline object_range::object_range(object& obj) noexcept {
+  auto b = obj.begin();
+  if (b.error()) { error_ = b.error(); return; }
+  begin_ = b.value_unsafe();
+  end_ = obj.end().value_unsafe();
+}
+
+simdjson_inline object_range_iterator object_range::begin() noexcept {
+  return object_range_iterator(begin_);
+}
+
+simdjson_inline object_range_iterator object_range::end() noexcept {
+  return object_range_iterator(end_);
+}
+
+//
+// Free functions
+//
+
+simdjson_inline array_range get_range(array& arr) noexcept {
+  return array_range(arr);
+}
+
+simdjson_inline object_range get_key_value_range(object& obj) noexcept {
+  return object_range(obj);
+}
+
+#if SIMDJSON_EXCEPTIONS
+simdjson_inline array_range get_range(simdjson_result<array> result) {
+  return array_range(result.value());
+}
+
+simdjson_inline object_range get_key_value_range(simdjson_result<object> result) {
+  return object_range(result.value());
+}
+#endif // SIMDJSON_EXCEPTIONS
+
+} // namespace ondemand
+} // namespace ppc64
+} // namespace simdjson
+
+// Verify the range wrapper types satisfy the expected C++20 concepts.
+static_assert(std::input_iterator<simdjson::ppc64::ondemand::array_range_iterator>);
+static_assert(std::input_iterator<simdjson::ppc64::ondemand::object_range_iterator>);
+static_assert(std::ranges::input_range<simdjson::ppc64::ondemand::array_range>);
+static_assert(std::ranges::input_range<simdjson::ppc64::ondemand::object_range>);
+static_assert(std::ranges::view<simdjson::ppc64::ondemand::array_range>);
+static_assert(std::ranges::view<simdjson::ppc64::ondemand::object_range>);
+
+#endif // SIMDJSON_SUPPORTS_RANGES
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H
+/* end file simdjson/generic/ondemand/ranges-inl.h for ppc64 */
 /* including simdjson/generic/ondemand/parser-inl.h for ppc64: #include "simdjson/generic/ondemand/parser-inl.h" */
 /* begin file simdjson/generic/ondemand/parser-inl.h for ppc64 */
 #ifndef SIMDJSON_GENERIC_ONDEMAND_PARSER_INL_H
@@ -127798,7 +163174,10 @@ simdjson_warn_unused simdjson_inline error_code parser::allocate(size_t new_capa

   // string_capacity copied from document::allocate
   _capacity = 0;
-  size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * new_capacity / 3 + SIMDJSON_PADDING, 64);
+  if(5 * (new_capacity / 3) + SIMDJSON_PADDING < SIMDJSON_PADDING) {
+    return CAPACITY; // overflow, only happen on legacy 32-bit systems with very large capacity
+  }
+  size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * (new_capacity / 3) + SIMDJSON_PADDING, 64);
   string_buf.reset(new (std::nothrow) uint8_t[string_capacity]);
 #if SIMDJSON_DEVELOPMENT_CHECKS
   start_positions.reset(new (std::nothrow) token_position[new_max_depth]);
@@ -127823,6 +163202,7 @@ simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate(p
   if (!json.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }

   json.remove_utf8_bom();
+  _document_len = json.length();

   // Allocate if needed
   if (capacity() < json.length() || !string_buf) {
@@ -127839,6 +163219,7 @@ simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate_a
   if (!json.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }

   json.remove_utf8_bom();
+  _document_len = json.length();

   // Allocate if needed
   if (capacity() < json.length() || !string_buf) {
@@ -127904,6 +163285,34 @@ simdjson_warn_unused simdjson_inline simdjson_result<json_iterator> parser::iter
   return json_iterator(reinterpret_cast<const uint8_t *>(json.data()), this);
 }

+inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size) noexcept {
+  if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
+  if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+    buf += 3;
+    len -= 3;
+  }
+  return document_stream(*this, buf, len, batch_size, false, stream_format::whitespace_delimited);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size) noexcept {
+  return iterate_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(padded_string_view s, size_t batch_size) noexcept {
+  if (!s.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
+  return iterate_many(s.data(), s.length(), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const padded_string &s, size_t batch_size) noexcept {
+  return iterate_many(padded_string_view(s), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const std::string &s, size_t batch_size) noexcept {
+  return iterate_many(padded_string_view(s), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size) noexcept {
+  return iterate_many(pad(s), batch_size);
+}
+
+#ifndef SIMDJSON_DISABLE_DEPRECATED_API
+SIMDJSON_PUSH_DISABLE_WARNINGS
+SIMDJSON_DISABLE_DEPRECATED_WARNING
 inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept {
   // Warning: no check is done on the buffer padding. We trust the user.
   if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
@@ -127911,8 +163320,11 @@ inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf,
     buf += 3;
     len -= 3;
   }
-  if(allow_comma_separated && batch_size < len) { batch_size = len; }
-  return document_stream(*this, buf, len, batch_size, allow_comma_separated);
+  // Map allow_comma_separated to stream_format::comma_delimited
+  if (allow_comma_separated) {
+    return document_stream(*this, buf, len, batch_size, false, stream_format::comma_delimited);
+  }
+  return document_stream(*this, buf, len, batch_size, false, stream_format::whitespace_delimited);
 }

 inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept {
@@ -127932,6 +163344,51 @@ inline simdjson_result<document_stream> parser::iterate_many(const std::string &
 inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size, bool allow_comma_separated) noexcept {
   return iterate_many(pad(s), batch_size, allow_comma_separated);
 }
+SIMDJSON_POP_DISABLE_WARNINGS
+#endif // SIMDJSON_DISABLE_DEPRECATED_API
+
+inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+  if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
+  if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+    buf += 3;
+    len -= 3;
+  }
+  if (format == stream_format::comma_delimited_array) {
+    // Strip leading JSON whitespace.
+    while (len > 0 && (buf[0] == ' ' || buf[0] == '\t' || buf[0] == '\n' || buf[0] == '\r')) {
+      buf++; len--;
+    }
+    // Expect the opening '['.
+    if (len == 0 || buf[0] != '[') { return TAPE_ERROR; }
+    buf++; len--;
+    // Strip trailing JSON whitespace.
+    while (len > 0 && (buf[len-1] == ' ' || buf[len-1] == '\t' || buf[len-1] == '\n' || buf[len-1] == '\r')) {
+      len--;
+    }
+    // Expect the closing ']'.
+    if (len == 0 || buf[len-1] != ']') { return TAPE_ERROR; }
+    len--;
+    // Fall through to comma_delimited over the array contents.
+    format = stream_format::comma_delimited;
+  }
+  return document_stream(*this, buf, len, batch_size, false, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+  return iterate_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(padded_string_view s, size_t batch_size, stream_format format) noexcept {
+  if (!s.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
+  return iterate_many(s.data(), s.length(), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const std::string &s, size_t batch_size, stream_format format) noexcept {
+  return iterate_many(padded_string_view(s), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size, stream_format format) noexcept {
+  return iterate_many(pad(s), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const padded_string &s, size_t batch_size, stream_format format) noexcept {
+  return iterate_many(padded_string_view(s), batch_size, format);
+}
 simdjson_pure simdjson_inline size_t parser::capacity() const noexcept {
   return _capacity;
 }
@@ -128339,6 +163796,27 @@ namespace simdjson {
 namespace ppc64 {
 namespace ondemand {

+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+simdjson_inline bool raw_json_string_is_quote_terminated(const uint8_t *json, uint32_t max_len) noexcept {
+  bool escaping{false};
+  for (uint32_t i = 1; i < max_len; i++) {
+    switch (json[i]) {
+      case '"':
+        if (!escaping) { return true; }
+        escaping = false;
+        break;
+      case '\\':
+        escaping = !escaping;
+        break;
+      default:
+        escaping = false;
+        break;
+    }
+  }
+  return false;
+}
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+
 simdjson_inline value_iterator::value_iterator(
   json_iterator *json_iter,
   depth_t depth,
@@ -128726,6 +164204,22 @@ simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iter
   return raw_json_string(key);
 }

+simdjson_warn_unused simdjson_inline error_code value_iterator::field_key_with_length(raw_json_string &key, std::size_t &len) noexcept {
+  assert_at_next();
+
+  const uint8_t *k = _json_iter->return_current_and_advance();
+  if (*(k++) != '"') { return report_error(TAPE_ERROR, "Object key is not a string"); }
+  // After return_current_and_advance(), the current token is the ':' that follows
+  // the key. The closing quote sits just before it (only JSON whitespace may
+  // intervene), so step back from the ':' to the closing quote to get the length.
+  // In minified JSON this is a single back-step.
+  const char *q = reinterpret_cast<const char *>(_json_iter->peek());
+  do { --q; } while (*q != '"');
+  key = raw_json_string(k);
+  len = static_cast<std::size_t>(q - reinterpret_cast<const char *>(k));
+  return SUCCESS;
+}
+
 simdjson_warn_unused simdjson_inline error_code value_iterator::field_value() noexcept {
   assert_at_next();

@@ -128843,7 +164337,7 @@ simdjson_warn_unused simdjson_inline error_code value_iterator::get_string(strin
   auto saved_string_buf_loc = _json_iter->string_buf_loc();
   auto err = get_string(allow_replacement).get(content);
   if (err) { return err; }
-  receiver = content;
+  internal::assign_utf8(receiver, content);
   // Restore the string buffer location, effectively discarding any temporary string storage
   _json_iter->string_buf_loc() = saved_string_buf_loc;
   return SUCCESS;
@@ -128854,6 +164348,15 @@ simdjson_warn_unused simdjson_inline simdjson_result<std::string_view> value_ite
 simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iterator::get_raw_json_string() noexcept {
   auto json = peek_scalar("string");
   if (*json != '"') { return incorrect_type_error("Not a string"); }
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  if (_json_iter->allow_incomplete_json()) {
+    const size_t remaining_input_length = _json_iter->remaining_input_length(json);
+    const uint32_t max_len = remaining_input_length < peek_start_length() ? uint32_t(remaining_input_length) : peek_start_length();
+    if (!raw_json_string_is_quote_terminated(json, max_len)) {
+      return STRING_ERROR;
+    }
+  }
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
   advance_scalar("string");
   return raw_json_string(json+1);
 }
@@ -128887,6 +164390,16 @@ simdjson_warn_unused simdjson_inline simdjson_result<double> value_iterator::get
   if(result.error() == SUCCESS) { advance_non_root_scalar("double"); }
   return result;
 }
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_float() noexcept {
+  auto result = numberparsing::parse_float(peek_non_root_scalar("float"));
+  if(result.error() == SUCCESS) { advance_non_root_scalar("float"); }
+  return result;
+}
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_float_in_string() noexcept {
+  auto result = numberparsing::parse_float_in_string(peek_non_root_scalar("float"));
+  if(result.error() == SUCCESS) { advance_non_root_scalar("float"); }
+  return result;
+}
 simdjson_warn_unused simdjson_inline simdjson_result<bool> value_iterator::get_bool() noexcept {
   auto result = parse_bool(peek_non_root_scalar("bool"));
   if(result.error() == SUCCESS) { advance_non_root_scalar("bool"); }
@@ -128989,7 +164502,7 @@ simdjson_warn_unused simdjson_inline error_code value_iterator::get_root_string(
   auto saved_string_buf_loc = _json_iter->string_buf_loc();
   auto err = get_root_string(check_trailing, allow_replacement).get(content);
   if (err) { return err; }
-  receiver = content;
+  internal::assign_utf8(receiver, content);
   // Restore the string buffer location, effectively discarding any temporary string storage
   _json_iter->string_buf_loc() = saved_string_buf_loc;
   return SUCCESS;
@@ -129001,6 +164514,15 @@ simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iter
   auto json = peek_scalar("string");
   if (*json != '"') { return incorrect_type_error("Not a string"); }
   if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  if (_json_iter->allow_incomplete_json()) {
+    const size_t remaining_input_length = _json_iter->remaining_input_length(json);
+    const uint32_t max_len = remaining_input_length < peek_root_length() ? uint32_t(remaining_input_length) : peek_root_length();
+    if (!raw_json_string_is_quote_terminated(json, max_len)) {
+      return STRING_ERROR;
+    }
+  }
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
   advance_scalar("string");
   return raw_json_string(json+1);
 }
@@ -129110,6 +164632,43 @@ simdjson_warn_unused simdjson_inline simdjson_result<double> value_iterator::get
   return result;
 }

+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_root_float(bool check_trailing) noexcept {
+  auto max_len = peek_root_length();
+  auto json = peek_root_scalar("float");
+  // We use the same buffer size as get_root_double: the number of significant
+  // digits that matter is smaller for binary32, but the JSON document may still
+  // spell out a long number that we must parse (and round) faithfully.
+  uint8_t tmpbuf[1074+8+1+1]; // +1 for null termination.
+  tmpbuf[1074+8+1] = '\0'; // make sure that buffer is always null terminated.
+  if (!_json_iter->copy_to_buffer(json, max_len, tmpbuf, 1074+8+1)) {
+    logger::log_error(*_json_iter, start_position(), depth(), "Root number more than 1082 characters");
+    return NUMBER_ERROR;
+  }
+  auto result = numberparsing::parse_float(tmpbuf);
+  if(result.error() == SUCCESS) {
+    if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+    advance_root_scalar("float");
+  }
+  return result;
+}
+
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_root_float_in_string(bool check_trailing) noexcept {
+  auto max_len = peek_root_length();
+  auto json = peek_root_scalar("float");
+  uint8_t tmpbuf[1074+8+1+1]; // +1 for null termination.
+  tmpbuf[1074+8+1] = '\0'; // make sure that buffer is always null terminated.
+  if (!_json_iter->copy_to_buffer(json, max_len, tmpbuf, 1074+8+1)) {
+    logger::log_error(*_json_iter, start_position(), depth(), "Root number more than 1082 characters");
+    return NUMBER_ERROR;
+  }
+  auto result = numberparsing::parse_float_in_string(tmpbuf);
+  if(result.error() == SUCCESS) {
+    if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+    advance_root_scalar("float");
+  }
+  return result;
+}
+
 simdjson_warn_unused simdjson_inline simdjson_result<bool> value_iterator::get_root_bool(bool check_trailing) noexcept {
   auto max_len = peek_root_length();
   auto json = peek_root_scalar("bool");
@@ -129338,6 +164897,21 @@ simdjson_inline void value_iterator::move_at_container_start() noexcept {
   _json_iter->token.set_position(_start_position + 1);
 }

+simdjson_inline void value_iterator::reenter_at(token_position position, depth_t depth) noexcept {
+  // Unlike reenter_child(), this does not require the live depth to be
+  // exactly one level shallower than depth, nor does it validate against
+  // the parser's per-depth container-start bookkeeping: neither holds in
+  // general for a caller-supplied snapshot (see object_position). What
+  // must still always hold, regardless of what was captured or how far
+  // the live iterator has since moved, is that position and depth are
+  // themselves sane values -- this is the same bound reenter_child()
+  // itself applies unconditionally.
+  SIMDJSON_ASSUME(position != nullptr);
+  SIMDJSON_ASSUME(depth >= 1 && depth < INT32_MAX);
+  _json_iter->_depth = depth;
+  _json_iter->token.set_position(position);
+}
+
 simdjson_inline simdjson_result<bool> value_iterator::reset_array() noexcept {
   if(error()) { return error(); }
   move_at_container_start();
@@ -130773,16 +166347,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
 }
 #endif

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
-                                uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
-  return _addcarry_u64(0, value1, value2,
-                       reinterpret_cast<unsigned __int64 *>(result));
-#else
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-#endif
-}

 } // unnamed namespace
 } // namespace westmere
@@ -131361,16 +166925,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
 }
 #endif

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
-                                uint64_t *result) {
-#if SIMDJSON_REGULAR_VISUAL_STUDIO
-  return _addcarry_u64(0, value1, value2,
-                       reinterpret_cast<unsigned __int64 *>(result));
-#else
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-#endif
-}

 } // unnamed namespace
 } // namespace westmere
@@ -131791,7 +167345,7 @@ simdjson_inline escaping escaping::copy_and_find(const uint8_t *src, uint8_t *ds
 /* end file simdjson/westmere/begin.h */
 /* including simdjson/generic/ondemand/amalgamated.h for westmere: #include "simdjson/generic/ondemand/amalgamated.h" */
 /* begin file simdjson/generic/ondemand/amalgamated.h for westmere */
-#if defined(SIMDJSON_CONDITIONAL_INCLUDE) && !defined(SIMDJSON_GENERIC_BUILDER_DEPENDENCIES_H)
+#if defined(SIMDJSON_CONDITIONAL_INCLUDE) && !defined(SIMDJSON_GENERIC_ONDEMAND_DEPENDENCIES_H)
 #error simdjson/generic/ondemand/dependencies.h must be included before simdjson/generic/ondemand/amalgamated.h!
 #endif

@@ -131840,6 +167394,13 @@ class token_iterator;
 class value;
 class value_iterator;

+#if SIMDJSON_SUPPORTS_RANGES
+class array_range;
+class array_range_iterator;
+class object_range;
+class object_range_iterator;
+#endif // SIMDJSON_SUPPORTS_RANGES
+
 } // namespace ondemand
 } // namespace westmere
 } // namespace simdjson
@@ -131872,6 +167433,9 @@ template <> struct is_builtin_deserializable<westmere::ondemand::object> : std::
 template <> struct is_builtin_deserializable<westmere::ondemand::value> : std::true_type {};
 template <> struct is_builtin_deserializable<westmere::ondemand::raw_json_string> : std::true_type {};
 template <> struct is_builtin_deserializable<std::string_view> : std::true_type {};
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template <> struct is_builtin_deserializable<std::u8string_view> : std::true_type {};
+#endif // SIMDJSON_SUPPORTS_CHAR8_T

 template <typename T>
 concept is_builtin_deserializable_v = is_builtin_deserializable<T>::value;
@@ -131889,6 +167453,10 @@ concept nothrow_custom_deserializable = nothrow_tag_invocable<deserialize_tag, V
 template <typename T, typename ValT = westmere::ondemand::value>
 concept nothrow_deserializable = nothrow_custom_deserializable<T, ValT> || is_builtin_deserializable_v<T>;

+// True iff get<T>() on ValT cannot throw: no custom tag_invoke, or that tag_invoke is noexcept.
+template <typename T, typename ValT = westmere::ondemand::value>
+concept nothrow_gettable = !custom_deserializable<T, ValT> || nothrow_custom_deserializable<T, ValT>;
+
 /// Deserialize Tag
 inline constexpr struct deserialize_tag {
   using array_type = westmere::ondemand::array;
@@ -132103,6 +167671,17 @@ public:
    */
   simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> field_key() noexcept;

+  /**
+   * Get the current field's key together with its raw byte length.
+   *
+   * Like field_key(), but also returns the number of raw key bytes (the distance
+   * from the first key byte to the closing quote). The length is recovered from
+   * the structural index -- the next structural token is the ':' -- by stepping
+   * back over any whitespace to the closing quote, avoiding a forward SIMD scan
+   * for the closing quote. Leaves the iterator positioned exactly as field_key().
+   */
+  simdjson_warn_unused simdjson_inline error_code field_key_with_length(raw_json_string &key, std::size_t &len) noexcept;
+
   /**
    * Pass the : in the field and move to its value.
    */
@@ -132255,6 +167834,8 @@ public:
   simdjson_warn_unused simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_warn_unused simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_warn_unused simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<bool> get_bool() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<bool> is_null() noexcept;
   simdjson_warn_unused simdjson_inline bool is_negative() noexcept;
@@ -132273,6 +167854,8 @@ public:
   simdjson_warn_unused simdjson_inline simdjson_result<int64_t> get_root_int64_in_string(bool check_trailing) noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<double> get_root_double(bool check_trailing) noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<double> get_root_double_in_string(bool check_trailing) noexcept;
+  simdjson_warn_unused simdjson_inline simdjson_result<float> get_root_float(bool check_trailing) noexcept;
+  simdjson_warn_unused simdjson_inline simdjson_result<float> get_root_float_in_string(bool check_trailing) noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<bool> get_root_bool(bool check_trailing) noexcept;
   simdjson_warn_unused simdjson_inline bool is_root_negative() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<bool> is_root_integer(bool check_trailing) noexcept;
@@ -132408,6 +167991,15 @@ protected:

   /** @copydoc error_code json_iterator::position() const noexcept; */
   simdjson_inline token_position position() const noexcept;
+  /**
+   * Move the live iterator directly to the given position and depth, without
+   * validating against the parser's per-depth container-start bookkeeping
+   * (unlike json_iterator::reenter_child()). Used to restore a previously
+   * captured mid-container position (see object::revert_position()): that
+   * bookkeeping only tracks each container's own start, not every position
+   * a caller might later capture and revert to, so it does not apply here.
+   */
+  simdjson_inline void reenter_at(token_position position, depth_t depth) noexcept;
   /** @copydoc error_code json_iterator::end_position() const noexcept; */
   simdjson_inline token_position last_position() const noexcept;
   /** @copydoc error_code json_iterator::end_position() const noexcept; */
@@ -132476,9 +168068,12 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+   * Supported types: object, array, raw_json_string, string_view, u8string_view (C++20), uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool
    *
-   * You may use get_double(), get_bool(), get_uint64(), get_int64(),
+   * You may use get_double(), get_float(), get_bool(), get_uint64(), get_int64(),
+   * get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(),
+   * get_float32(), get_float64(),
    * get_object(), get_array(), get_raw_json_string(), or get_string() instead.
    * When SIMDJSON_SUPPORTS_CONCEPTS is set, custom types are also supported.
    *
@@ -132488,7 +168083,7 @@ public:
   template <typename T>
   simdjson_inline simdjson_result<T> get()
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+    noexcept(nothrow_gettable<T, value>)
 #else
     noexcept
 #endif
@@ -132503,7 +168098,8 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+   * Supported types: object, array, raw_json_string, string_view, u8string_view (C++20), uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool
    * If the macro SIMDJSON_SUPPORTS_CONCEPTS is set, then custom types are also supported.
    *
    * @param out This is set to a value of the given type, parsed from the JSON. If there is an error, this may not be initialized.
@@ -132513,7 +168109,7 @@ public:
   template <typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out)
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+    noexcept(nothrow_gettable<T, value>)
 #else
     noexcept
 #endif
@@ -132541,7 +168137,7 @@ public:
       "And you do not seem to have added support for it. Indeed, we have that "
       "simdjson::custom_deserializable<T> is false and the type T is not a default type "
       "such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-      "int64_t, double, or bool.");
+      "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
     static_cast<void>(out); // to get rid of unused errors
     return UNINITIALIZED;
   }
@@ -132550,7 +168146,8 @@ public:
     // immediately fail.
     static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
       "The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-      "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+      "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+      "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
       " get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
       " You may also add support for custom types, see our documentation.");
     static_cast<void>(out); // to get rid of unused errors
@@ -132628,6 +168225,50 @@ public:
    */
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;

+  /**
+   * Cast this JSON value to a 16-bit unsigned integer.
+   *
+   * Calls get_uint64() and checks that the result fits in a uint16_t.
+   *
+   * @returns A 16-bit unsigned integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint16_t.
+   */
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+
+  /**
+   * Cast this JSON value to a 16-bit signed integer.
+   *
+   * Calls get_int64() and checks that the result fits in an int16_t.
+   *
+   * @returns A 16-bit signed integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int16_t.
+   */
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+
+  /**
+   * Cast this JSON value to an 8-bit unsigned integer.
+   *
+   * Calls get_uint64() and checks that the result fits in a uint8_t.
+   *
+   * @returns An 8-bit unsigned integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint8_t.
+   */
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+
+  /**
+   * Cast this JSON value to an 8-bit signed integer.
+   *
+   * Calls get_int64() and checks that the result fits in an int8_t.
+   *
+   * @returns An 8-bit signed integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int8_t.
+   */
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
+
   /**
    * Cast this JSON value to a double.
    *
@@ -132644,6 +168285,53 @@ public:
    */
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;

+  /**
+   * Cast this JSON value to a float (binary32).
+   *
+   * The value is rounded directly to binary32: it is the float nearest to the
+   * JSON number. Note that this may differ from get_double() followed by a cast
+   * to float, which rounds twice.
+   *
+   * @returns A float.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+
+  /**
+   * Cast this JSON value (inside string) to a float (binary32).
+   *
+   * @returns A float.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  /**
+   * Cast this JSON value to a std::float32_t (C++23).
+   *
+   * Same as get_float(): the value is rounded directly to binary32.
+   *
+   * @returns A std::float32_t.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  /**
+   * Cast this JSON value to a std::float64_t (C++23).
+   *
+   * Same as get_double().
+   *
+   * @returns A std::float64_t.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   */
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
+
   /**
    * Cast this JSON value to a string.
    *
@@ -132671,6 +168359,26 @@ public:
    */
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;

+#if SIMDJSON_SUPPORTS_CHAR8_T
+  /**
+   * Cast this JSON value to a C++20 UTF-8 string.
+   *
+   * The string is guaranteed to be valid UTF-8.
+   *
+   * Equivalent to get<std::u8string_view>(). This is a zero-copy alias of
+   * get_string(): the very same bytes are returned, viewed as char8_t.
+   *
+   * Important: a value should be consumed once. Calling get_u8string() twice on the same
+   * value is an error.
+   *
+   * @param allow_replacement Whether to allow a replacement character for unmatched surrogate pairs.
+   * @returns An UTF-8 string. The string is stored in the parser and will be invalidated the next
+   *          time it parses a document or when it is destroyed.
+   * @returns INCORRECT_TYPE if the JSON value is not a string.
+   */
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
   /**
    * Attempts to fill the provided std::string reference with the parsed value of the current string.
    *
@@ -132758,7 +168466,7 @@ public:
   /**
    * Cast this JSON value to an unsigned integer.
    *
-   * @returns A signed 64-bit integer.
+   * @returns A unsigned 64-bit integer.
    * @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit unsigned integer.
    */
   simdjson_inline operator uint64_t() noexcept(false);
@@ -133223,9 +168931,24 @@ public:
   simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template <typename string_type>
   simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -133233,9 +168956,19 @@ public:
   simdjson_inline simdjson_result<bool> get_bool() noexcept;
   simdjson_inline simdjson_result<bool> is_null() noexcept;

-  template<typename T> simdjson_inline simdjson_result<T> get() noexcept;
+  template<typename T> simdjson_inline simdjson_result<T> get()
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, westmere::ondemand::value>);
+#else
+    noexcept;
+#endif

-  template<typename T> simdjson_inline error_code get(T &out) noexcept;
+  template<typename T> simdjson_inline error_code get(T &out)
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, westmere::ondemand::value>);
+#else
+    noexcept;
+#endif

 #if SIMDJSON_EXCEPTIONS
   template <class T>
@@ -133566,6 +169299,7 @@ protected:
   token_position _position{};

   friend class json_iterator;
+  friend class document_stream;
   friend class value_iterator;
   friend class object;
   template <typename... Args>
@@ -133657,6 +169391,9 @@ protected:
    * value of this attribute.
    */
   bool _streaming{false};
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  bool _allow_incomplete_json{false};
+#endif

 public:
   simdjson_inline json_iterator() noexcept = default;
@@ -133681,6 +169418,10 @@ public:
    * start_root_array() and start_root_object().
    */
   simdjson_inline bool streaming() const noexcept;
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  simdjson_inline bool allow_incomplete_json() const noexcept;
+  simdjson_inline size_t remaining_input_length(const uint8_t *json) const noexcept;
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON

   /**
    * Get the root value iterator
@@ -134560,33 +170301,87 @@ public:
    * @param batch_size The batch size to use. MUST be larger than the largest document. The sweet
    *                   spot is cache-related: small enough to fit in cache, yet big enough to
    *                   parse as many documents as possible in one tight loop.
-   *                   Defaults to 10MB, which has been a reasonable sweet spot in our tests.
-   * @param allow_comma_separated (defaults on false) This allows a mode where the documents are
-   *                   separated by commas instead of whitespace. It comes with a performance
-   *                   penalty because the entire document is indexed at once (and the document must be
-   *                   less than 4 GB), and there is no multithreading. In this mode, the batch_size parameter
-   *                   is effectively ignored, as it is set to at least the document size.
+   *                   Defaults to 1MB, which has been a reasonable sweet spot in our tests.
+   * @param allow_comma_separated @deprecated Use stream_format::comma_delimited instead.
+   *                   When true, maps internally to stream_format::comma_delimited.
+   *                   Defaults to false.
    * @return The stream, or an error. An empty input will yield 0 documents rather than an EMPTY error. Errors:
    *         - MEMALLOC if the parser does not have enough capacity and memory allocation fails
    *         - CAPACITY if the parser does not have enough capacity and batch_size > max_capacity.
    *         - other json errors if parsing fails. You should not rely on these errors to always the same for the
    *           same document: they may vary under runtime dispatch (so they may vary depending on your system and hardware).
    */
-  inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
-  inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
-  inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
-  inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size)
+  inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size)
     the string might be automatically padded with up to SIMDJSON_PADDING whitespace characters */
-  inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
-  inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
+  inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @private An rvalue input is destroyed at the end of the full-expression, while the
+   * returned document_stream only holds a pointer to it: iterating the stream would then
+   * read freed memory. These deleted overloads also catch a std::string_view argument,
+   * which would otherwise convert implicitly to a padded_string temporary. */
+  inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size = DEFAULT_BATCH_SIZE) = delete;// unsafe
+  /** @private @overload iterate_many(const std::string &&s, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size = DEFAULT_BATCH_SIZE) = delete;// unsafe
   /** @private We do not want to allow implicit conversion from C string to std::string. */
   simdjson_result<document_stream> iterate_many(const char *buf, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept = delete;

+#ifndef SIMDJSON_DISABLE_DEPRECATED_API
+  /**
+   * @deprecated Use iterate_many with stream_format::comma_delimited instead.
+   */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @private Deleted for the same reason as iterate_many(const std::string &&s, size_t batch_size). */
+  inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size, bool allow_comma_separated) = delete;// unsafe
+  /** @private @overload iterate_many(const std::string &&s, size_t batch_size, bool allow_comma_separated) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size, bool allow_comma_separated) = delete;// unsafe
+#endif // SIMDJSON_DISABLE_DEPRECATED_API
+
+  /**
+   * Parse a stream of JSON documents with explicit format specification.
+   *
+   * @param buf The concatenated JSON documents.
+   * @param len The length of the buffer.
+   * @param batch_size The batch size to use.
+   * @param format The stream format (whitespace_delimited, json_sequence, or comma_delimited).
+   * @return A stream of documents, or an error.
+   */
+  inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(padded_string_view s, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format)
+   *
+   * The string is padded in place if needed (see simdjson::pad), as with iterate_many(std::string &s, size_t batch_size).
+   */
+  inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size, stream_format format) noexcept;
+  /** @private Deleted for the same reason as iterate_many(const std::string &&s, size_t batch_size). */
+  inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+  /** @private @overload iterate_many(const std::string &&s, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+
   /** The capacity of this parser (the largest document it can process). */
   simdjson_pure simdjson_inline size_t capacity() const noexcept;
   /** The maximum capacity of this parser (the largest document it is allowed to process). */
@@ -134714,6 +170509,7 @@ private:
   size_t _capacity{0};
   size_t _max_capacity;
   size_t _max_depth{DEFAULT_MAX_DEPTH};
+  size_t _document_len{0};
   std::unique_ptr<uint8_t[]> string_buf{};

 #if SIMDJSON_DEVELOPMENT_CHECKS
@@ -134776,8 +170572,19 @@ public:
    * Begin array iteration.
    *
    * Part of the std::iterable interface.
+   *
+   * With SIMDJSON_DEVELOPMENT_CHECKS, the iterator locks this array while it is
+   * alive, so that reentrant access (at(), count_elements(), reset(), ...) is
+   * reported as OUT_OF_ORDER_ITERATION.
    */
-  simdjson_inline simdjson_result<array_iterator> begin() noexcept;
+  simdjson_inline simdjson_result<array_iterator> begin() & noexcept;
+  /**
+   * Begin iteration over a temporary array, e.g., `v.get_array().begin()`.
+   *
+   * The iterator does not depend on the array instance and may outlive it, so
+   * it does not lock it.
+   */
+  simdjson_inline simdjson_result<array_iterator> begin() && noexcept;
   /**
    * Sentinel representing the end of the array.
    *
@@ -134908,7 +170715,7 @@ public:
    */
   template <typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out)
-     noexcept(custom_deserializable<T, array> ? nothrow_custom_deserializable<T, array> : true) {
+     noexcept(nothrow_gettable<T, array>) {
     static_assert(custom_deserializable<T, array>);
     return deserialize(*this, out);
   }
@@ -134920,7 +170727,7 @@ public:
    */
   template <typename T>
   simdjson_inline simdjson_result<T> get()
-    noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+    noexcept(nothrow_gettable<T, array>)
   {
     static_assert(std::is_default_constructible<T>::value, "The specified type is not default constructible.");
     T out{};
@@ -134976,6 +170783,10 @@ protected:
    * iter.is_alive() == false indicates iteration is complete.
    */
   value_iterator iter{};
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  bool locked{false};
+  simdjson_inline void set_locked(bool _locked) noexcept;
+#endif

   friend class value;
   friend class document;
@@ -134997,7 +170808,8 @@ public:
   simdjson_inline simdjson_result(error_code error) noexcept; ///< @private
   simdjson_inline simdjson_result() noexcept = default;

-  simdjson_inline simdjson_result<westmere::ondemand::array_iterator> begin() noexcept;
+  simdjson_inline simdjson_result<westmere::ondemand::array_iterator> begin() & noexcept;
+  simdjson_inline simdjson_result<westmere::ondemand::array_iterator> begin() && noexcept;
   simdjson_inline simdjson_result<westmere::ondemand::array_iterator> end() noexcept;
   inline simdjson_result<size_t> count_elements() & noexcept;
   inline simdjson_result<bool> is_empty() & noexcept;
@@ -135017,7 +170829,7 @@ public:
   // TODO: move this code into object-inl.h

   template<typename T>
-  simdjson_inline simdjson_result<T> get() noexcept {
+  simdjson_inline simdjson_result<T> get() noexcept(nothrow_gettable<T, westmere::ondemand::array>) {
     if (error()) { return error(); }
     if constexpr (std::is_same_v<T, westmere::ondemand::array>) {
       return first;
@@ -135025,7 +170837,7 @@ public:
     return first.get<T>();
   }
   template<typename T>
-  simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept {
+  simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept(nothrow_gettable<T, westmere::ondemand::array>) {
     if (error()) { return error(); }
     if constexpr (std::is_same_v<T, westmere::ondemand::array>) {
       out = first;
@@ -135077,6 +170889,15 @@ public:
   /** Create a new, invalid array iterator. */
   simdjson_inline array_iterator() noexcept = default;

+#if SIMDJSON_DEVELOPMENT_CHECKS
+  simdjson_inline ~array_iterator() noexcept;
+
+  simdjson_inline array_iterator(array_iterator&&) noexcept;
+  simdjson_inline array_iterator& operator=(array_iterator&&) noexcept;
+  simdjson_inline array_iterator(const array_iterator&) noexcept;
+  simdjson_inline array_iterator& operator=(const array_iterator&) noexcept;
+#endif
+
   //
   // Iterator interface
   //
@@ -135119,6 +170940,9 @@ public:
 private:
 #if SIMDJSON_DEVELOPMENT_CHECKS
    bool has_been_referenced{false};
+   array* parent{nullptr};
+
+   simdjson_inline array_iterator(const value_iterator &_iter, array* _parent) noexcept;
 #endif
   value_iterator iter{};

@@ -135218,14 +171042,14 @@ public:
   /**
    * Cast this JSON value to an unsigned integer.
    *
-   * @returns A signed 64-bit integer.
+   * @returns A unsigned 64-bit integer.
    * @returns INCORRECT_TYPE If the JSON value is not a 64-bit unsigned integer.
    */
   simdjson_inline simdjson_result<uint64_t> get_uint64() noexcept;
   /**
    * Cast this JSON value (inside string) to an unsigned integer.
    *
-   * @returns A signed 64-bit integer.
+   * @returns A unsigned 64-bit integer.
    * @returns INCORRECT_TYPE If the JSON value is not a 64-bit unsigned integer.
    */
   simdjson_inline simdjson_result<uint64_t> get_uint64_in_string() noexcept;
@@ -135263,6 +171087,46 @@ public:
    * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int32_t.
    */
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  /**
+   * Cast this JSON value to a 16-bit unsigned integer.
+   *
+   * Calls get_uint64() and checks that the result fits in a uint16_t.
+   *
+   * @returns A 16-bit unsigned integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint16_t.
+   */
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  /**
+   * Cast this JSON value to a 16-bit signed integer.
+   *
+   * Calls get_int64() and checks that the result fits in an int16_t.
+   *
+   * @returns A 16-bit signed integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int16_t.
+   */
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  /**
+   * Cast this JSON value to an 8-bit unsigned integer.
+   *
+   * Calls get_uint64() and checks that the result fits in a uint8_t.
+   *
+   * @returns An 8-bit unsigned integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint8_t.
+   */
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  /**
+   * Cast this JSON value to an 8-bit signed integer.
+   *
+   * Calls get_int64() and checks that the result fits in an int8_t.
+   *
+   * @returns An 8-bit signed integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int8_t.
+   */
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   /**
    * Cast this JSON value to a double.
    *
@@ -135278,6 +171142,53 @@ public:
    * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
    */
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+
+  /**
+   * Cast this JSON value to a float (binary32).
+   *
+   * The value is rounded directly to binary32: it is the float nearest to the
+   * JSON number. Note that this may differ from get_double() followed by a cast
+   * to float, which rounds twice.
+   *
+   * @returns A float.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+
+  /**
+   * Cast this JSON value (inside string) to a float (binary32).
+   *
+   * @returns A float.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  /**
+   * Cast this JSON value to a std::float32_t (C++23).
+   *
+   * Same as get_float(): the value is rounded directly to binary32.
+   *
+   * @returns A std::float32_t.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  /**
+   * Cast this JSON value to a std::float64_t (C++23).
+   *
+   * Same as get_double().
+   *
+   * @returns A std::float64_t.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   */
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   /**
    * Cast this JSON value to a string.
    *
@@ -135291,6 +171202,24 @@ public:
    * @returns INCORRECT_TYPE if the JSON value is not a string.
    */
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  /**
+   * Cast this JSON value to a C++20 UTF-8 string.
+   *
+   * The string is guaranteed to be valid UTF-8.
+   *
+   * Equivalent to get<std::u8string_view>(). This is a zero-copy alias of
+   * get_string(): the very same bytes are returned, viewed as char8_t.
+   *
+   * Important: Calling get_u8string() twice on the same document is an error.
+   *
+   * @param allow_replacement Whether to allow a replacement character for unmatched surrogate pairs.
+   * @returns An UTF-8 string. The string is stored in the parser and will be invalidated the next
+   *          time it parses a document or when it is destroyed.
+   * @returns INCORRECT_TYPE if the JSON value is not a string.
+   */
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   /**
    * Attempts to fill the provided std::string reference with the parsed value of the current string.
    *
@@ -135361,9 +171290,12 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool
    *
-   * You may use get_double(), get_bool(), get_uint64(), get_int64(),
+   * You may use get_double(), get_float(), get_bool(), get_uint64(), get_int64(),
+   * get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(),
+   * get_float32(), get_float64(),
    * get_object(), get_array(), get_raw_json_string(), or get_string() instead.
    *
    * @returns A value of the given type, parsed from the JSON.
@@ -135372,7 +171304,7 @@ public:
   template <typename T>
   simdjson_inline simdjson_result<T> get() &
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    noexcept(nothrow_gettable<T, document>)
 #else
     noexcept
 #endif
@@ -135395,7 +171327,7 @@ public:
   template<typename T>
   simdjson_inline simdjson_result<T> get() &&
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    noexcept(nothrow_gettable<T, document>)
 #else
     noexcept
 #endif
@@ -135407,7 +171339,8 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool, value
+   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool, value
    *
    * Be mindful that the document instance must remain in scope while you are accessing object, array and value instances.
    *
@@ -135418,7 +171351,7 @@ public:
   template<typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out) &
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    noexcept(nothrow_gettable<T, document>)
 #else
     noexcept
 #endif
@@ -135431,7 +171364,7 @@ public:
         "And you do not seem to have added support for it. Indeed, we have that "
         "simdjson::custom_deserializable<T> is false and the type T is not a default type "
         "such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-        "int64_t, double, or bool.");
+        "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
       static_cast<void>(out); // to get rid of unused errors
       return UNINITIALIZED;
     }
@@ -135440,7 +171373,8 @@ public:
     // immediately fail.
     static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
       "The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-      "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+      "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+      "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
       " get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
       " You may also add support for custom types, see our documentation.");
     static_cast<void>(out); // to get rid of unused errors
@@ -135449,7 +171383,12 @@ public:
   }

   /** @overload template<typename T> error_code get(T &out) & noexcept */
-  template<typename T> simdjson_deprecated simdjson_inline error_code get(T &out) && noexcept;
+  template<typename T> simdjson_deprecated simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, document>);
+#else
+    noexcept;
+#endif

 #if SIMDJSON_EXCEPTIONS
   /**
@@ -135483,24 +171422,24 @@ public:
   /**
    * Cast this JSON value to an unsigned integer.
    *
-   * @returns A signed 64-bit integer.
+   * @returns A unsigned 64-bit integer.
    * @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit unsigned integer.
    */
-  simdjson_inline operator uint64_t() noexcept(false);
+  explicit simdjson_inline operator uint64_t() noexcept(false);
   /**
    * Cast this JSON value to a signed integer.
    *
    * @returns A signed 64-bit integer.
    * @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit integer.
    */
-  simdjson_inline operator int64_t() noexcept(false);
+  explicit simdjson_inline operator int64_t() noexcept(false);
   /**
    * Cast this JSON value to a double.
    *
    * @returns A double.
    * @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a valid floating-point number.
    */
-  simdjson_inline operator double() noexcept(false);
+  explicit simdjson_inline operator double() noexcept(false);
   /**
    * Cast this JSON value to a string.
    *
@@ -135510,7 +171449,7 @@ public:
    *          time it parses a document or when it is destroyed.
    * @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not a string.
    */
-  simdjson_inline operator std::string_view() noexcept(false) simdjson_lifetime_bound;
+  explicit simdjson_inline operator std::string_view() noexcept(false) simdjson_lifetime_bound;
   /**
    * Cast this JSON value to a raw_json_string.
    *
@@ -135519,14 +171458,14 @@ public:
    * @returns A pointer to the raw JSON for the given string.
    * @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not a string.
    */
-  simdjson_inline operator raw_json_string() noexcept(false) simdjson_lifetime_bound;
+  explicit simdjson_inline operator raw_json_string() noexcept(false) simdjson_lifetime_bound;
   /**
    * Cast this JSON value to a bool.
    *
    * @returns A bool value.
    * @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not true or false.
    */
-  simdjson_inline operator bool() noexcept(false);
+  explicit simdjson_inline operator bool() noexcept(false);
   /**
    * Cast this JSON value to a value when the document is an object or an array.
    *
@@ -136021,9 +171960,24 @@ public:
   simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template <typename string_type>
   simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -136035,7 +171989,7 @@ public:
   template <typename T>
   simdjson_inline simdjson_result<T> get() &
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    noexcept(nothrow_gettable<T, document_reference>)
 #else
     noexcept
 #endif
@@ -136048,7 +172002,8 @@ public:
   template<typename T>
   simdjson_inline simdjson_result<T> get() &&
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    // Forwards to document::get<T>(), so the document customization decides.
+    noexcept(nothrow_gettable<T, document>)
 #else
     noexcept
 #endif
@@ -136060,7 +172015,8 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool, value
+   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool, value
    *
    * Be mindful that the document instance must remain in scope while you are accessing object, array and value instances.
    *
@@ -136071,7 +172027,7 @@ public:
   template<typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out) &
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document_reference> : true)
+    noexcept(nothrow_gettable<T, document_reference>)
 #else
     noexcept
 #endif
@@ -136084,7 +172040,7 @@ public:
         "And you do not seem to have added support for it. Indeed, we have that "
         "simdjson::custom_deserializable<T> is false and the type T is not a default type "
         "such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-        "int64_t, double, or bool.");
+        "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
       static_cast<void>(out); // to get rid of unused errors
       return UNINITIALIZED;
     }
@@ -136093,7 +172049,8 @@ public:
     // immediately fail.
     static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
       "The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-      "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+      "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+      "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
       " get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
       " You may also add support for custom types, see our documentation.");
     static_cast<void>(out); // to get rid of unused errors
@@ -136102,7 +172059,12 @@ public:
   }

   /** @overload template<typename T> error_code get(T &out) & noexcept */
-  template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+  template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, document_reference>);
+#else
+    noexcept;
+#endif
   simdjson_inline simdjson_result<std::string_view> raw_json() noexcept;
 #if SIMDJSON_STATIC_REFLECTION
   template<constevalutil::fixed_string... FieldNames, typename T>
@@ -136115,12 +172077,12 @@ public:
   explicit simdjson_inline operator T() noexcept(false);
   simdjson_inline operator array() & noexcept(false);
   simdjson_inline operator object() & noexcept(false);
-  simdjson_inline operator uint64_t() noexcept(false);
-  simdjson_inline operator int64_t() noexcept(false);
-  simdjson_inline operator double() noexcept(false);
-  simdjson_inline operator std::string_view() noexcept(false);
-  simdjson_inline operator raw_json_string() noexcept(false);
-  simdjson_inline operator bool() noexcept(false);
+  explicit simdjson_inline operator uint64_t() noexcept(false);
+  explicit simdjson_inline operator int64_t() noexcept(false);
+  explicit simdjson_inline operator double() noexcept(false);
+  explicit simdjson_inline operator std::string_view() noexcept(false);
+  explicit simdjson_inline operator raw_json_string() noexcept(false);
+  explicit simdjson_inline operator bool() noexcept(false);
   simdjson_inline operator value() noexcept(false);
 #endif
   simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -136182,9 +172144,24 @@ public:
   simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template <typename string_type>
   simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -136193,11 +172170,31 @@ public:
   simdjson_inline simdjson_result<westmere::ondemand::value> get_value() noexcept;
   simdjson_inline simdjson_result<bool> is_null() noexcept;

-  template<typename T> simdjson_inline simdjson_result<T> get() & noexcept;
-  template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() && noexcept;
+  template<typename T> simdjson_inline simdjson_result<T> get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, westmere::ondemand::document>);
+#else
+    noexcept;
+#endif
+  template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, westmere::ondemand::document>);
+#else
+    noexcept;
+#endif

-  template<typename T> simdjson_inline error_code get(T &out) & noexcept;
-  template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+  template<typename T> simdjson_inline error_code get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, westmere::ondemand::document>);
+#else
+    noexcept;
+#endif
+  template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, westmere::ondemand::document>);
+#else
+    noexcept;
+#endif
 #if SIMDJSON_EXCEPTIONS

   using westmere::implementation_simdjson_result_base<westmere::ondemand::document>::operator*;
@@ -136206,12 +172203,12 @@ public:
   explicit simdjson_inline operator T() noexcept(false);
   simdjson_inline operator westmere::ondemand::array() & noexcept(false);
   simdjson_inline operator westmere::ondemand::object() & noexcept(false);
-  simdjson_inline operator uint64_t() noexcept(false);
-  simdjson_inline operator int64_t() noexcept(false);
-  simdjson_inline operator double() noexcept(false);
-  simdjson_inline operator std::string_view() noexcept(false);
-  simdjson_inline operator westmere::ondemand::raw_json_string() noexcept(false);
-  simdjson_inline operator bool() noexcept(false);
+  explicit simdjson_inline operator uint64_t() noexcept(false);
+  explicit simdjson_inline operator int64_t() noexcept(false);
+  explicit simdjson_inline operator double() noexcept(false);
+  explicit simdjson_inline operator std::string_view() noexcept(false);
+  explicit simdjson_inline operator westmere::ondemand::raw_json_string() noexcept(false);
+  explicit simdjson_inline operator bool() noexcept(false);
   simdjson_inline operator westmere::ondemand::value() noexcept(false);
 #endif
   simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -136277,9 +172274,24 @@ public:
   simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template <typename string_type>
   simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -136288,22 +172300,42 @@ public:
   simdjson_inline simdjson_result<westmere::ondemand::value> get_value() noexcept;
   simdjson_inline simdjson_result<bool> is_null() noexcept;

-  template<typename T> simdjson_inline simdjson_result<T> get() & noexcept;
-  template<typename T> simdjson_inline simdjson_result<T> get() && noexcept;
+  template<typename T> simdjson_inline simdjson_result<T> get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, westmere::ondemand::document_reference>);
+#else
+    noexcept;
+#endif
+  template<typename T> simdjson_inline simdjson_result<T> get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, westmere::ondemand::document_reference>);
+#else
+    noexcept;
+#endif

-  template<typename T> simdjson_inline error_code get(T &out) & noexcept;
-  template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+  template<typename T> simdjson_inline error_code get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, westmere::ondemand::document_reference>);
+#else
+    noexcept;
+#endif
+  template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, westmere::ondemand::document_reference>);
+#else
+    noexcept;
+#endif
 #if SIMDJSON_EXCEPTIONS
   template <class T>
   explicit simdjson_inline operator T() noexcept(false);
   simdjson_inline operator westmere::ondemand::array() & noexcept(false);
   simdjson_inline operator westmere::ondemand::object() & noexcept(false);
-  simdjson_inline operator uint64_t() noexcept(false);
-  simdjson_inline operator int64_t() noexcept(false);
-  simdjson_inline operator double() noexcept(false);
-  simdjson_inline operator std::string_view() noexcept(false);
-  simdjson_inline operator westmere::ondemand::raw_json_string() noexcept(false);
-  simdjson_inline operator bool() noexcept(false);
+  explicit simdjson_inline operator uint64_t() noexcept(false);
+  explicit simdjson_inline operator int64_t() noexcept(false);
+  explicit simdjson_inline operator double() noexcept(false);
+  explicit simdjson_inline operator std::string_view() noexcept(false);
+  explicit simdjson_inline operator westmere::ondemand::raw_json_string() noexcept(false);
+  explicit simdjson_inline operator bool() noexcept(false);
   simdjson_inline operator westmere::ondemand::value() noexcept(false);
 #endif
   simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -136471,10 +172503,7 @@ public:
    *   }
    *   size_t truncated = stream.truncated_bytes();
    *
-   * IMPORTANT: this value is only meaningful under the conditions below. It is
-   * computed from stage-1 bookkeeping, and outside these conditions it is not
-   * merely imprecise, it is arbitrary -- it can exceed size_in_bytes() or wrap
-   * around to a huge value. Check it only when both of the following hold:
+   * IMPORTANT: this value is only meaningful under the conditions below.
    *
    *   - you iterated all the way to the end of the stream;
    *   - no document reported an error. Iteration stops at the first failed
@@ -136483,6 +172512,9 @@ public:
    * If you need to know about a truncated tail outside those conditions, track
    * it yourself from the last successful document (see iterator::current_index()
    * and iterator::source()).
+   *
+   * An empty input (zero bytes) or an input made only of white space contains
+   * no document: truncated_bytes() returns zero.
    */
   inline size_t truncated_bytes() const noexcept;

@@ -136542,7 +172574,10 @@ public:
      *
      * The returned string_view instance is simply a map to the (unparsed)
      * source string: it may thus include white-space characters and all manner
-     * of padding.
+     * of padding. It spans the whole current document, whether or not you
+     * have already accessed (part of) the document. Thus
+     * current_index() + source().size() is the offset just past the end of the
+     * current document, which is useful when reading a stream in chunks.
      *
      * This function (source()) is experimental and the usage
      * may change in future versions of simdjson: we find the API somewhat
@@ -136596,13 +172631,16 @@ private:
    * @param buf is the raw byte buffer we need to process
    * @param len is the length of the raw byte buffer in bytes
    * @param batch_size is the size of the windows (must be strictly greater or equal to the largest JSON document)
+   * @param allow_comma_separated whether to allow comma-separated documents
+   * @param format the stream format
    */
   simdjson_inline document_stream(
     ondemand::parser &parser,
     const uint8_t *buf,
     size_t len,
     size_t batch_size,
-    bool allow_comma_separated
+    bool allow_comma_separated,
+    stream_format format = stream_format::whitespace_delimited
   ) noexcept;

   /**
@@ -136636,8 +172674,23 @@ private:
    */
   inline void next() noexcept;

-  /** Move the json_iterator of the document to the location of the next document in the stream. */
+  /**
+   * Move the json_iterator of the document to the location of the next document
+   * in the stream.
+   *
+   * For formats with a document delimiter (`newline_delimited`, `json_sequence`),
+   * when the iterator is still inside the current document (`depth() > 0`), this
+   * may jump to the next delimiter instead of walking remaining structurals. That
+   * jump does not structure-validate the unread remainder.
+   */
   inline void next_document() noexcept;
+  /** Byte that ends a document under `format`, or 0 if there is none. */
+  simdjson_inline uint8_t document_delimiter() const noexcept;
+  /**
+   * Position the iterator at the first structural at or past the next
+   * `delimiter` in the current batch. Returns false if none is found.
+   */
+  simdjson_inline bool skip_to_delimiter(uint8_t delimiter) noexcept;

   /** Get the next document index. */
   inline size_t next_batch_start() const noexcept;
@@ -136651,6 +172704,7 @@ private:
   size_t len;
   size_t batch_size;
   bool allow_comma_separated;
+  stream_format format;
   /**
    * We are going to use just one document instance. The document owns
    * the json_iterator. It implies that we only ever pass a reference
@@ -136677,7 +172731,7 @@ private:
   /** The error returned from the stage 1 thread. */
   error_code stage1_thread_error{UNINITIALIZED};
   /** The thread used to run stage 1 against the next batch in the background. */
-  std::unique_ptr<stage1_worker> worker{new(std::nothrow) stage1_worker()};
+  std::unique_ptr<stage1_worker> worker{};
   /**
    * The parser used to run stage 1 in the background. Will be swapped
    * with the regular parser when finished.
@@ -136752,6 +172806,16 @@ public:
    * call it again nor can you call key().
    */
   simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  /**
+   * Get the key as a C++20 u8string_view. This is a zero-copy alias of
+   * unescaped_key(): the very same bytes are returned, viewed as char8_t.
+   *
+   * This consumes the key: once you have called unescaped_u8key(), you cannot
+   * call it again nor can you call key().
+   */
+  simdjson_inline simdjson_warn_unused simdjson_result<std::u8string_view> unescaped_u8key(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   /**
    * Get the key as a string_view (for higher speed, consider raw_key).
    * We deliberately use a more cumbersome name (unescaped_key) to force users
@@ -136784,6 +172848,16 @@ public:
    * you can safely call it repeatedly. Use unescaped_key() to get the unescaped key.
    */
   simdjson_inline std::string_view escaped_key() const noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  /**
+   * Get the key as a C++20 u8string_view. This is a zero-copy alias of
+   * escaped_key(): the very same bytes are returned, viewed as char8_t.
+   * The string is unprocessed, so it may contain escape characters
+   * (e.g., \uXXXX or \n). It does not count as a consumption of the content:
+   * you can safely call it repeatedly.
+   */
+  simdjson_inline std::u8string_view escaped_u8key() const noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   /**
    * Get the field value.
    */
@@ -136815,11 +172889,17 @@ public:
   simdjson_inline simdjson_result() noexcept = default;

   simdjson_inline simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> unescaped_u8key(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template<typename string_type>
   simdjson_inline error_code unescaped_key(string_type &receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<westmere::ondemand::raw_json_string> key() noexcept;
   simdjson_inline simdjson_result<std::string_view> key_raw_json_token() noexcept;
   simdjson_inline simdjson_result<std::string_view> escaped_key() noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> escaped_u8key() noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   simdjson_inline simdjson_result<westmere::ondemand::value> value() noexcept;
 };

@@ -136827,6 +172907,1398 @@ public:

 #endif // SIMDJSON_GENERIC_ONDEMAND_FIELD_H
 /* end file simdjson/generic/ondemand/field.h for westmere */
+/* including simdjson/generic/ondemand/key_selector.h for westmere: #include "simdjson/generic/ondemand/key_selector.h" */
+/* begin file simdjson/generic/ondemand/key_selector.h for westmere */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+#define SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #include "simdjson/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/common_defs.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/constevalutil.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/raw_json_string.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#include <array>
+#include <string>      // std::string (key_selector::describe)
+#include <string_view>
+#include <cstddef>
+#include <cstdint>
+#include <cstring>     // std::memcpy (portable unaligned window load)
+#include <utility>     // std::index_sequence (window candidate dispatch)
+#include <type_traits> // std::integral_constant (window candidate dispatch)
+
+// AArch64 only: 32-bit ARM (ARMv7) defines __ARM_NEON too, but lacks the
+// A64-only horizontal reductions (vmaxvq_u32) used below. See issue #2885.
+#if defined(__aarch64__)
+  #include <arm_neon.h>
+  #define SIMDJSON_KEY_SELECTOR_HAS_NEON 1
+#else
+  #define SIMDJSON_KEY_SELECTOR_HAS_NEON 0
+#endif
+#if defined(__SSE2__)
+  #include <emmintrin.h>
+  #define SIMDJSON_KEY_SELECTOR_HAS_SSE2 1
+#else
+  #define SIMDJSON_KEY_SELECTOR_HAS_SSE2 0
+#endif
+#if defined(__loongarch_sx)
+  #include <lsxintrin.h>
+  #define SIMDJSON_KEY_SELECTOR_HAS_LSX 1
+#else
+  #define SIMDJSON_KEY_SELECTOR_HAS_LSX 0
+#endif
+
+#if SIMDJSON_SUPPORTS_CONCEPTS
+
+namespace simdjson {
+namespace westmere {
+namespace ondemand {
+
+namespace key_selector_detail {
+
+// Since not constexpr, triggers compile-time error.
+inline void compile_time_error(const char* message) noexcept { (void)message; }
+
+// ============================================================================
+// Compile-time perfect-hash generator.
+// It scales to ~100 keys at compile time by determining association values one (position, character)
+// symbol at a time (gperf-style) instead of an exhaustive offset search, and
+// falls back to a Hash-and-Displace construction for large/awkward key sets.
+//
+// Only flat tables survive to runtime; the lookup is a few additions plus a
+// single SIMD key comparison (see match_raw below).
+// ============================================================================
+
+// Maximum number of character positions the gperf hash may combine.
+static constexpr std::size_t MAX_POSITIONS = 16;
+// Sentinel "position" meaning "the last character of the key".
+static constexpr std::size_t LAST_CHAR = std::size_t(-1);
+// Runtime-encoded sentinels (stored in uint8 tables).
+static constexpr std::uint8_t POS_LAST_CHAR = 0xFF; // positions_[i] == last char
+static constexpr std::uint8_t HD_MODE       = 0xFF; // num_positions == H&D mode
+// Flags stored in positions[2] in H&D mode to select the key-hash variant.
+static constexpr std::size_t HD_HASH_2BYTE_FLAG = 2;
+static constexpr std::size_t HD_HASH_4BYTE_FLAG = 4;
+
+constexpr std::size_t next_power_of_2(std::size_t n) noexcept {
+    if (n == 0) { return 1; }
+    std::size_t p = 1;
+    while (p < n) { p <<= 1; }
+    return p;
+}
+
+// Character at a given position (LAST_CHAR means last character), or 256 if out
+// of bounds.
+constexpr std::size_t char_at(std::string_view key, std::size_t pos) noexcept {
+    if (pos == LAST_CHAR) {
+        if (key.empty()) { return 256; }
+        return static_cast<unsigned char>(key[key.size() - 1]);
+    }
+    if (pos >= key.size()) { return 256; }
+    return static_cast<unsigned char>(key[pos]);
+}
+
+// Count key pairs that a set of positions fails to distinguish. Keys whose
+// lengths differ modulo the table size are separated by the length term in the
+// hash, so they need no position coverage.
+template <std::size_t N>
+consteval std::size_t count_undistinguished_pairs(
+    const std::array<std::string_view, N>& keys,
+    const std::size_t* positions,
+    std::size_t num_positions,
+    std::size_t modulus) {
+    std::size_t count = 0;
+    for (std::size_t i = 0; i < N; ++i) {
+        for (std::size_t j = i + 1; j < N; ++j) {
+            if (keys[i].size() % modulus != keys[j].size() % modulus) { continue; }
+            bool distinguished = false;
+            for (std::size_t p = 0; p < num_positions; ++p) {
+                if (char_at(keys[i], positions[p]) != char_at(keys[j], positions[p])) {
+                    distinguished = true;
+                    break;
+                }
+            }
+            if (!distinguished) { ++count; }
+        }
+    }
+    return count;
+}
+
+template <std::size_t N>
+consteval bool positions_distinguish(
+    const std::array<std::string_view, N>& keys,
+    const std::size_t* positions,
+    std::size_t num_positions,
+    std::size_t modulus) {
+    return count_undistinguished_pairs<N>(keys, positions, num_positions, modulus) == 0;
+}
+
+// Number of distinct (length % modulus, char_at(key, pos)) pairs at a position.
+template <std::size_t N>
+consteval std::size_t discriminating_power(
+    const std::array<std::string_view, N>& keys,
+    std::size_t pos,
+    std::size_t modulus) {
+    struct pair { std::size_t len_mod; std::size_t ch; };
+    std::array<pair, N> pairs{};
+    for (std::size_t i = 0; i < N; ++i) {
+        pairs[i] = {keys[i].size() % modulus, char_at(keys[i], pos)};
+    }
+    std::size_t count = 0;
+    for (std::size_t i = 0; i < N; ++i) {
+        bool dup = false;
+        for (std::size_t j = 0; j < i; ++j) {
+            if (pairs[i].len_mod == pairs[j].len_mod && pairs[i].ch == pairs[j].ch) {
+                dup = true;
+                break;
+            }
+        }
+        if (!dup) { ++count; }
+    }
+    return count;
+}
+
+template <std::size_t N>
+consteval std::size_t max_key_length(const std::array<std::string_view, N>& keys) {
+    std::size_t m = 0;
+    for (std::size_t i = 0; i < N; ++i) {
+        if (keys[i].size() > m) { m = keys[i].size(); }
+    }
+    return m;
+}
+
+// Bounded backtracking DFS for a minimal set of distinguishing positions.
+template <std::size_t N>
+consteval bool backtracking_search(
+    const std::array<std::string_view, N>& keys,
+    const std::size_t* candidates,
+    std::size_t num_candidates,
+    std::size_t* positions,
+    std::size_t& num_positions_out,
+    std::size_t& budget,
+    std::size_t modulus) {
+    constexpr std::size_t MAX_DEPTH = 8;
+    std::size_t breadth = num_candidates < 20 ? num_candidates : 20;
+
+    struct frame { std::size_t depth; std::size_t next_ci; std::size_t parent_count; };
+    std::array<frame, MAX_DEPTH + 1> stack{};
+    std::size_t sp = 0;
+
+    std::size_t initial_count = count_undistinguished_pairs<N>(keys, positions, 0, modulus);
+    if (budget > 0) { --budget; }
+    if (initial_count == 0) { num_positions_out = 0; return true; }
+
+    stack[0] = {0, 0, initial_count};
+
+    while (budget > 0) {
+        if (sp > MAX_DEPTH) {
+            if (sp == 0) { break; }
+            --sp;
+            ++stack[sp].next_ci;
+            continue;
+        }
+        auto& f = stack[sp];
+        if (f.next_ci >= breadth) {
+            if (sp == 0) { break; }
+            --sp;
+            ++stack[sp].next_ci;
+            continue;
+        }
+        positions[sp] = candidates[f.next_ci];
+        --budget;
+        std::size_t new_count = count_undistinguished_pairs<N>(keys, positions, sp + 1, modulus);
+        if (new_count == 0) { num_positions_out = sp + 1; return true; }
+        if (new_count < f.parent_count && sp + 1 < MAX_DEPTH) {
+            stack[sp + 1] = {sp + 1, f.next_ci + 1, new_count};
+            ++sp;
+        } else {
+            ++f.next_ci;
+        }
+    }
+    return false;
+}
+
+// Phase 1: select character positions that distinguish all colliding pairs.
+template <std::size_t N>
+consteval std::size_t select_positions(
+    const std::array<std::string_view, N>& keys,
+    std::array<std::size_t, MAX_POSITIONS>& positions,
+    std::size_t modulus) {
+    if (positions_distinguish<N>(keys, positions.data(), 0, modulus)) { return 0; }
+
+    std::size_t max_len = max_key_length(keys);
+    constexpr std::size_t MAX_CANDIDATES = 256;
+    std::array<std::size_t, MAX_CANDIDATES> candidates{};
+    std::array<std::size_t, MAX_CANDIDATES> powers{};
+    std::size_t num_candidates = 0;
+    for (std::size_t p = 0; p < max_len && num_candidates < MAX_CANDIDATES - 1; ++p) {
+        candidates[num_candidates] = p;
+        powers[num_candidates] = discriminating_power(keys, p, modulus);
+        ++num_candidates;
+    }
+    if (num_candidates < MAX_CANDIDATES) {
+        candidates[num_candidates] = LAST_CHAR;
+        powers[num_candidates] = discriminating_power(keys, LAST_CHAR, modulus);
+        ++num_candidates;
+    }
+    for (std::size_t i = 0; i < num_candidates; ++i) {
+        for (std::size_t j = i + 1; j < num_candidates; ++j) {
+            if (powers[j] > powers[i]) {
+                auto tc = candidates[i]; candidates[i] = candidates[j]; candidates[j] = tc;
+                auto tp = powers[i]; powers[i] = powers[j]; powers[j] = tp;
+            }
+        }
+    }
+
+    positions[0] = candidates[0];
+    if (positions_distinguish<N>(keys, positions.data(), 1, modulus)) { return 1; }
+
+    positions[0] = 0;
+    positions[1] = LAST_CHAR;
+    if (positions_distinguish<N>(keys, positions.data(), 2, modulus)) { return 2; }
+
+    {
+        std::size_t budget = 5000;
+        std::size_t num_found = 0;
+        if (backtracking_search<N>(keys, candidates.data(), num_candidates,
+                                   positions.data(), num_found, budget, modulus)) {
+            return num_found;
+        }
+    }
+
+    std::size_t num_pos = 0;
+    for (std::size_t ci = 0; ci < num_candidates && num_pos < MAX_POSITIONS; ++ci) {
+        bool already = false;
+        for (std::size_t p = 0; p < num_pos; ++p) {
+            if (positions[p] == candidates[ci]) { already = true; break; }
+        }
+        if (already) { continue; }
+        positions[num_pos] = candidates[ci];
+        ++num_pos;
+        if (positions_distinguish<N>(keys, positions.data(), num_pos, modulus)) { return num_pos; }
+    }
+
+    compile_time_error("Failed to find distinguishing positions for perfect hash");
+    return 0;
+}
+
+// Result of PHF computation. A max-sized slot_to_key array lets the same struct
+// type carry any chosen table size.
+template <std::size_t N>
+struct phf_result {
+    // Allow up to 8x the minimum table size. Sparser tables solve faster.
+    static constexpr std::size_t MAX_TABLE_SIZE = next_power_of_2(N) * 8;
+    std::size_t table_size{};
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso_values{};
+    std::size_t num_positions{};
+    std::array<std::size_t, MAX_POSITIONS> positions{};
+    std::array<std::size_t, MAX_TABLE_SIZE> slot_to_key{};
+};
+
+// Partition-based asso_values search (gperf-style). Determines asso_values one
+// (position, character) symbol at a time; never revisits a value. Equivalence
+// classes (keys sharing the same undetermined symbols) keep the search cheap.
+template <std::size_t N, std::size_t M>
+consteval bool try_generate_gperf(
+    const std::array<std::string_view, N>& keys,
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS>& asso_values,
+    std::size_t& num_positions,
+    std::array<std::size_t, MAX_POSITIONS>& positions,
+    std::array<std::size_t, M>& slot_to_key) {
+    num_positions = select_positions<N>(keys, positions, M);
+
+    for (std::size_t p = 0; p < MAX_POSITIONS; ++p) {
+        for (std::size_t c = 0; c < 256; ++c) { asso_values[p][c] = 0; }
+    }
+
+    if (num_positions == 0) {
+        for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+        for (std::size_t i = 0; i < N; ++i) {
+            std::size_t slot = keys[i].size() % M;
+            if (slot_to_key[slot] != N) { return false; }
+            slot_to_key[slot] = i;
+        }
+        return true;
+    }
+
+    std::array<std::array<std::size_t, MAX_POSITIONS>, N> kchars{};
+    for (std::size_t k = 0; k < N; ++k) {
+        for (std::size_t p = 0; p < num_positions; ++p) {
+            kchars[k][p] = char_at(keys[k], positions[p]);
+        }
+    }
+
+    struct sym_t { std::size_t pos; std::size_t ch; std::size_t freq; };
+    constexpr std::size_t MAX_SYMS = MAX_POSITIONS * 256;
+    std::array<sym_t, MAX_SYMS> syms{};
+    std::size_t nsyms = 0;
+    for (std::size_t p = 0; p < num_positions; ++p) {
+        std::array<std::size_t, 256> freq{};
+        for (std::size_t k = 0; k < N; ++k) {
+            std::size_t c = kchars[k][p];
+            if (c < 256) { freq[c]++; }
+        }
+        for (std::size_t c = 0; c < 256; ++c) {
+            if (freq[c] > 0) { syms[nsyms++] = {p, c, freq[c]}; }
+        }
+    }
+    for (std::size_t i = 0; i < nsyms; ++i) {
+        for (std::size_t j = i + 1; j < nsyms; ++j) {
+            if (syms[j].freq > syms[i].freq) {
+                auto tmp = syms[i]; syms[i] = syms[j]; syms[j] = tmp;
+            }
+        }
+    }
+
+    std::array<std::size_t, N> phash{};
+    for (std::size_t k = 0; k < N; ++k) { phash[k] = keys[k].size(); }
+
+    std::array<std::array<uint64_t, 256>, MAX_POSITIONS> salt{};
+    {
+        uint64_t s = 0x9e3779b97f4a7c15ULL;
+        for (std::size_t p = 0; p < num_positions; ++p) {
+            for (std::size_t c = 0; c < 256; ++c) {
+                s = s * 6364136223846793005ULL + 1442695040888963407ULL;
+                salt[p][c] = s;
+            }
+        }
+    }
+    std::array<uint64_t, N> sig{};
+    for (std::size_t k = 0; k < N; ++k) {
+        uint64_t s = 0;
+        for (std::size_t p = 0; p < num_positions; ++p) {
+            std::size_t c = kchars[k][p];
+            if (c < 256) { s ^= salt[p][c]; }
+        }
+        sig[k] = s;
+    }
+    std::array<std::size_t, N> order{};
+    for (std::size_t k = 0; k < N; ++k) { order[k] = k; }
+
+    std::array<std::size_t, M> slot_gen{};
+    std::size_t gen = 0;
+
+    std::size_t search_limit = next_power_of_2(M);
+    if (search_limit < 32) { search_limit = 32; }
+
+    for (std::size_t si = 0; si < nsyms; ++si) {
+        std::size_t sp = syms[si].pos;
+        std::size_t sc = syms[si].ch;
+
+        uint64_t sp_salt = salt[sp][sc];
+        for (std::size_t k = 0; k < N; ++k) {
+            if (kchars[k][sp] == sc) { sig[k] ^= sp_salt; }
+        }
+
+        for (std::size_t i = 1; i < N; ++i) {
+            std::size_t x = order[i];
+            uint64_t xs = sig[x];
+            std::size_t j = i;
+            while (j > 0 && sig[order[j - 1]] > xs) {
+                order[j] = order[j - 1];
+                --j;
+            }
+            order[j] = x;
+        }
+
+        bool found = false;
+        for (std::size_t v = 0; v < search_limit && !found; ++v) {
+            bool collision = false;
+            std::size_t ci = 0;
+            while (ci < N && !collision) {
+                uint64_t class_sig = sig[order[ci]];
+                std::size_t cj = ci;
+                while (cj < N && sig[order[cj]] == class_sig) { ++cj; }
+                if (cj - ci > 1) {
+                    ++gen;
+                    for (std::size_t x = ci; x < cj; ++x) {
+                        std::size_t k = order[x];
+                        std::size_t h = phash[k];
+                        if (kchars[k][sp] == sc) { h += v; }
+                        h %= M;
+                        if (slot_gen[h] == gen) { collision = true; break; }
+                        slot_gen[h] = gen;
+                    }
+                }
+                ci = cj;
+            }
+            if (!collision) {
+                asso_values[sp][sc] = v;
+                for (std::size_t k = 0; k < N; ++k) {
+                    if (kchars[k][sp] == sc) { phash[k] += v; }
+                }
+                found = true;
+            }
+        }
+        if (!found) { return false; }
+    }
+
+    for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+    for (std::size_t i = 0; i < N; ++i) {
+        std::size_t slot = phash[i] % M;
+        if (slot_to_key[slot] != N) { return false; }
+        slot_to_key[slot] = i;
+    }
+    std::size_t filled = 0;
+    for (std::size_t i = 0; i < M; ++i) {
+        if (slot_to_key[i] != N) { ++filled; }
+    }
+    return filled == N;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_compute_phf(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+    static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso{};
+    std::size_t npos{};
+    std::array<std::size_t, MAX_POSITIONS> pos{};
+    std::array<std::size_t, M> s2k{};
+    if (try_generate_gperf<N, M>(keys, asso, npos, pos, s2k)) {
+        result.table_size = M;
+        result.asso_values = asso;
+        result.num_positions = npos;
+        result.positions = pos;
+        for (std::size_t i = 0; i < M; ++i) { result.slot_to_key[i] = s2k[i]; }
+        for (std::size_t i = M; i < phf_result<N>::MAX_TABLE_SIZE; ++i) { result.slot_to_key[i] = N; }
+        return true;
+    }
+    return false;
+}
+
+template <std::size_t N, std::size_t M, std::size_t MaxM>
+consteval bool try_gperf_po2(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+    if (try_compute_phf<N, M>(keys, result)) { return true; }
+    constexpr std::size_t NextM = M * 2;
+    if constexpr (NextM <= MaxM) { return try_gperf_po2<N, NextM, MaxM>(keys, result); }
+    return false;
+}
+
+// --- Hash-and-Displace fallback --------------------------------------------
+
+constexpr std::size_t hd_bucket_hash(std::string_view key) noexcept {
+    std::size_t c0 = key.empty() ? 0 : static_cast<unsigned char>(key[0]);
+    std::size_t c1 = key.empty() ? 0 : static_cast<unsigned char>(key[key.size() - 1]);
+    return (c0 + c1 * 3 + key.size() * 17) & 0xFF;
+}
+constexpr std::size_t hd_safe_char(const char* p, std::size_t len, std::size_t idx) noexcept {
+    std::size_t has = static_cast<std::size_t>(idx < len);
+    std::size_t si = idx & (std::size_t{0} - has);
+    return static_cast<unsigned char>(p[si]) & (std::size_t{0} - has);
+}
+constexpr std::size_t hd_key_hash_2(std::string_view key) noexcept {
+    std::size_t kc = key.size();
+    kc = kc * 31 + static_cast<unsigned char>(key[0]);
+    kc = kc * 31 + hd_safe_char(key.data(), key.size(), 1);
+    return kc;
+}
+constexpr std::size_t hd_key_hash_4(std::string_view key) noexcept {
+    std::size_t kc = key.size();
+    kc = kc * 31 + static_cast<unsigned char>(key[0]);
+    kc = kc * 31 + hd_safe_char(key.data(), key.size(), 1);
+    kc = kc * 31 + hd_safe_char(key.data(), key.size(), 2);
+    kc = kc * 31 + hd_safe_char(key.data(), key.size(), 3);
+    return kc;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_hash_and_displace(
+    const std::array<std::string_view, N>& keys,
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS>& asso_values,
+    std::size_t& num_positions,
+    std::array<std::size_t, MAX_POSITIONS>& positions,
+    std::array<std::size_t, M>& slot_to_key) {
+    num_positions = select_positions<N>(keys, positions, M);
+
+    if (num_positions == 0) {
+        for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+        for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+        for (std::size_t i = 0; i < N; ++i) {
+            std::size_t slot = keys[i].size() % M;
+            if (slot_to_key[slot] != N) { return false; }
+            slot_to_key[slot] = i;
+        }
+        return true;
+    }
+
+    for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+    num_positions = HD_MODE; // sentinel for H&D mode
+    positions[0] = 0;
+    positions[1] = LAST_CHAR;
+
+    std::array<std::size_t, N> key_bucket{};
+    for (std::size_t i = 0; i < N; ++i) { key_bucket[i] = hd_bucket_hash(keys[i]); }
+
+    struct bucket_info { std::size_t ch; std::size_t count; };
+    std::array<bucket_info, N> buckets{};
+    std::size_t num_buckets = 0;
+    for (std::size_t i = 0; i < N; ++i) {
+        std::size_t bk = key_bucket[i];
+        bool found = false;
+        for (std::size_t b = 0; b < num_buckets; ++b) {
+            if (buckets[b].ch == bk) { ++buckets[b].count; found = true; break; }
+        }
+        if (!found) { buckets[num_buckets++] = {bk, 1}; }
+    }
+    for (std::size_t i = 0; i < num_buckets; ++i) {
+        for (std::size_t j = i + 1; j < num_buckets; ++j) {
+            if (buckets[j].count > buckets[i].count) {
+                auto tmp = buckets[i]; buckets[i] = buckets[j]; buckets[j] = tmp;
+            }
+        }
+    }
+
+    auto try_placement = [&](auto key_hash_fn) -> bool {
+        for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+        for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+        for (std::size_t b = 0; b < num_buckets; ++b) {
+            std::size_t ch = buckets[b].ch;
+            std::array<std::size_t, N> bucket_keys{};
+            std::size_t bk_count = 0;
+            for (std::size_t i = 0; i < N; ++i) {
+                if (key_bucket[i] == ch) { bucket_keys[bk_count++] = i; }
+            }
+            bool placed = false;
+            std::size_t max_d = M < 255 ? M : 255;
+            for (std::size_t d = 0; d < max_d; ++d) {
+                bool ok = true;
+                std::array<std::size_t, N> bucket_slots{};
+                for (std::size_t k = 0; k < bk_count; ++k) {
+                    std::size_t slot = (d + key_hash_fn(keys[bucket_keys[k]])) % M;
+                    if (slot_to_key[slot] != N) { ok = false; break; }
+                    for (std::size_t k2 = 0; k2 < k; ++k2) {
+                        if (bucket_slots[k2] == slot) { ok = false; break; }
+                    }
+                    if (!ok) { break; }
+                    bucket_slots[k] = slot;
+                }
+                if (ok) {
+                    asso_values[0][ch] = d;
+                    for (std::size_t k = 0; k < bk_count; ++k) {
+                        slot_to_key[bucket_slots[k]] = bucket_keys[k];
+                    }
+                    placed = true;
+                    break;
+                }
+            }
+            if (!placed) { return false; }
+        }
+        std::size_t filled = 0;
+        for (std::size_t i = 0; i < M; ++i) {
+            if (slot_to_key[i] != N) { ++filled; }
+        }
+        return filled == N;
+    };
+
+    if (try_placement([](std::string_view k) { return hd_key_hash_2(k); })) {
+        positions[2] = HD_HASH_2BYTE_FLAG;
+        return true;
+    }
+    if (try_placement([](std::string_view k) { return hd_key_hash_4(k); })) {
+        positions[2] = HD_HASH_4BYTE_FLAG;
+        return true;
+    }
+    return false;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_compute_phf_hd(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+    static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso{};
+    std::size_t npos{};
+    std::array<std::size_t, MAX_POSITIONS> pos{};
+    std::array<std::size_t, M> s2k{};
+    if (try_hash_and_displace<N, M>(keys, asso, npos, pos, s2k)) {
+        result.table_size = M;
+        result.asso_values = asso;
+        result.num_positions = npos;
+        result.positions = pos;
+        for (std::size_t i = 0; i < M; ++i) { result.slot_to_key[i] = s2k[i]; }
+        for (std::size_t i = M; i < phf_result<N>::MAX_TABLE_SIZE; ++i) { result.slot_to_key[i] = N; }
+        return true;
+    }
+    return false;
+}
+
+template <std::size_t N, std::size_t M>
+consteval phf_result<N> compute_phf_hd_po2(const std::array<std::string_view, N>& keys) {
+    static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+    phf_result<N> result{};
+    if (try_compute_phf_hd<N, M>(keys, result)) { return result; }
+    constexpr std::size_t NextM = M * 2;
+    if constexpr (NextM <= phf_result<N>::MAX_TABLE_SIZE) {
+        return compute_phf_hd_po2<N, NextM>(keys);
+    } else {
+        compile_time_error("Hash-and-Displace: failed to find valid table size");
+        return result;
+    }
+}
+
+// Compute a perfect hash for `keys`: try gperf at power-of-two sizes (capped so
+// the runtime tables stay within uint8 indices), then fall back to H&D.
+template <std::size_t N>
+consteval phf_result<N> compute_phf(const std::array<std::string_view, N>& keys) {
+    constexpr std::size_t StartM = next_power_of_2(N);
+    constexpr std::size_t GPERF_MAX_TABLE =
+        phf_result<N>::MAX_TABLE_SIZE < 256 ? phf_result<N>::MAX_TABLE_SIZE : 256;
+    if constexpr (StartM <= GPERF_MAX_TABLE) {
+        phf_result<N> result{};
+        if (try_gperf_po2<N, StartM, GPERF_MAX_TABLE>(keys, result)) { return result; }
+    }
+    return compute_phf_hd_po2<N, StartM>(keys);
+}
+
+// ============================================================================
+// Runtime tables (flat, uint8) derived from a phf_result.
+// ============================================================================
+
+template <std::size_t N, std::size_t TableSize, std::size_t MaxKeyLen>
+struct phf_data {
+    std::array<std::array<std::uint8_t, 256>, MAX_POSITIONS> asso_values{};
+    std::array<std::uint8_t, MAX_POSITIONS>                  positions{};
+    std::uint8_t                                             num_positions{};
+    std::uint8_t                                             hd_hash_variant{}; // 2 or 4 (H&D only)
+    std::array<std::uint8_t, TableSize>                      slot_to_key{};
+    // slot_key_bytes[s] holds the key stored at slot s, zero-padded to a 16-byte
+    // multiple so the SIMD comparison can read a whole register.
+    std::array<std::array<char, ((MaxKeyLen + 15) / 16) * 16>, TableSize> slot_key_bytes{};
+    std::array<std::uint8_t, TableSize>                      slot_key_len{};
+};
+
+template <std::size_t N>
+constexpr std::size_t compute_max_key_len(const std::array<std::string_view, N>& keys) noexcept {
+    std::size_t m = 0;
+    for (std::size_t i = 0; i < N; ++i) { if (keys[i].size() > m) { m = keys[i].size(); } }
+    return m;
+}
+
+// Validate keys and build the runtime tables from the computed perfect hash.
+template <std::size_t N, std::size_t TableSize, std::size_t MaxKeyLen>
+consteval phf_data<N, TableSize, MaxKeyLen>
+build_phf_data(const std::array<std::string_view, N>& keys, const phf_result<N>& result) {
+    for (std::size_t i = 0; i < N; ++i) {
+        if (keys[i].empty())            { compile_time_error("empty keys are not allowed in key_selector"); }
+        if (keys[i].size() > MaxKeyLen) { compile_time_error("key length exceeds MaxKeyLen"); }
+        for (char c : keys[i]) {
+            if (c == '\\') { compile_time_error("backslash not allowed in key_selector keys"); }
+            if (c == '"')  { compile_time_error("quote not allowed in key_selector keys"); }
+            if (c == '\0') { compile_time_error("null byte not allowed in key_selector keys"); }
+        }
+        for (std::size_t j = i + 1; j < N; ++j) {
+            if (keys[i] == keys[j]) { compile_time_error("duplicate keys in key_selector"); }
+        }
+    }
+
+    phf_data<N, TableSize, MaxKeyLen> out{};
+
+    if (result.num_positions == HD_MODE) {
+        // H&D mode: single displacement table in asso_values[0].
+        for (std::size_t c = 0; c < 256; ++c) {
+            out.asso_values[0][c] = static_cast<std::uint8_t>(result.asso_values[0][c]);
+        }
+        out.num_positions   = static_cast<std::uint8_t>(HD_MODE);
+        out.hd_hash_variant = static_cast<std::uint8_t>(result.positions[2]);
+    } else {
+        for (std::size_t pi = 0; pi < result.num_positions; ++pi) {
+            for (std::size_t c = 0; c < 256; ++c) {
+                out.asso_values[pi][c] = static_cast<std::uint8_t>(result.asso_values[pi][c] % TableSize);
+            }
+        }
+        out.num_positions = static_cast<std::uint8_t>(result.num_positions);
+        for (std::size_t i = 0; i < result.num_positions; ++i) {
+            out.positions[i] = (result.positions[i] == LAST_CHAR)
+                ? POS_LAST_CHAR
+                : static_cast<std::uint8_t>(result.positions[i]);
+        }
+    }
+
+    for (std::size_t s = 0; s < TableSize; ++s) {
+        out.slot_to_key[s] = static_cast<std::uint8_t>(result.slot_to_key[s]);
+    }
+
+    for (std::size_t s = 0; s < TableSize; ++s) {
+        std::size_t ki = result.slot_to_key[s];
+        if (ki < N) {
+            auto k = keys[ki];
+            out.slot_key_len[s] = static_cast<std::uint8_t>(k.size());
+            for (std::size_t c = 0; c < k.size(); ++c) { out.slot_key_bytes[s][c] = k[c]; }
+        } else {
+            out.slot_key_len[s] = 0; // empty slot: no length can match
+        }
+    }
+    return out;
+}
+
+// --- SIMD runtime primitives ------------------------------------------------
+
+// The runtime matchers read whole 16-byte blocks up to offset 63 (four blocks)
+// for keys as long as the 63-character maximum. That read must stay within the
+// buffer's trailing padding, so the padding has to exceed the largest offset we
+// touch.
+static_assert(SIMDJSON_PADDING > 63,
+              "key_selector requires SIMDJSON_PADDING > 63 for its SIMD key reads");
+
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+// True when every byte of `diff` is zero. A 32-bit-lane horizontal max is
+// enough (zero iff every 32-bit word is zero) and is cheaper than a byte-wide
+// reduction. Do not replace this with a floating-point compare against 0.0:
+// flush-to-zero would treat a denormal as zero.
+simdjson_really_inline bool neon_all_bytes_zero(uint8x16_t diff) noexcept {
+    return vmaxvq_u32(vreinterpretq_u32_u8(diff)) == 0;
+}
+#endif
+
+// Scan for the terminating '"' starting at p. Returns its byte offset (= key
+// length). Caller guarantees SIMDJSON_PADDING (== 64) bytes past the JSON buffer,
+// so reading whole 16-byte blocks up to offset 63 is always safe.
+template <std::size_t MaxKeyLen>
+simdjson_really_inline std::size_t scan_key_length(const char* p) noexcept {
+    // The closing quote of the longest key sits at most at offset MaxKeyLen, so we
+    // scan as many 16-byte blocks as it takes to cover offsets 0..MaxKeyLen. The
+    // cap (63) keeps every covered offset within the 64-byte padding guarantee, so
+    // the SIMD and scalar builds agree.
+    static_assert(MaxKeyLen <= 63, "MaxKeyLen must be <= 63 for current SIMD implementations");
+    // [[maybe_unused]]: the scalar fallback (no NEON/SSE2/LSX) does not use it.
+    [[maybe_unused]] constexpr std::size_t num_blocks = MaxKeyLen / 16 + 1;
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+    for (std::size_t b = 0; b < num_blocks; ++b) {
+        uint8x16_t v = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + b * 16);
+        uint8x16_t cmp = vceqq_u8(v, vdupq_n_u8('"'));
+        uint64_t m = vget_lane_u64(
+            vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(cmp), 4)), 0);
+        if (simdjson_likely(m != 0)) { return b * 16 + (std::size_t(__builtin_ctzll(m)) >> 2); }
+    }
+    return MaxKeyLen + 1;
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+    for (std::size_t b = 0; b < num_blocks; ++b) {
+        __m128i v   = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + b * 16));
+        __m128i cmp = _mm_cmpeq_epi8(v, _mm_set1_epi8('"'));
+        unsigned m  = static_cast<unsigned>(_mm_movemask_epi8(cmp));
+        if (simdjson_likely(m != 0)) { return b * 16 + std::size_t(__builtin_ctz(m)); }
+    }
+    return MaxKeyLen + 1;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+    for (std::size_t b = 0; b < num_blocks; ++b) {
+        __m128i v   = __lsx_vld(reinterpret_cast<const void*>(p + b * 16), 0);
+        __m128i cmp = __lsx_vseq_b(v, __lsx_vreplgr2vr_b('"'));
+        // vmskltz_b gathers the per-byte sign bits (set where the byte equals '"')
+        // into the low 16 bits of lane 0, the LSX equivalent of movemask.
+        unsigned m  = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(cmp), 0)) & 0xFFFFu;
+        if (simdjson_likely(m != 0)) { return b * 16 + std::size_t(__builtin_ctz(m)); }
+    }
+    return MaxKeyLen + 1;
+#else
+    for (std::size_t i = 0; i <= MaxKeyLen; ++i)
+        if (p[i] == '"') return i;
+    return MaxKeyLen + 1;
+#endif
+}
+
+// Byte-equal of p[0..len) against stored[0..len). stored is zero-padded past
+// `len`. Input is read over 16, 32, 48, or 64 bytes (padded JSON buffer
+// guaranteed; SIMDJSON_PADDING == 64).
+template <std::size_t MaxKeyLen>
+simdjson_really_inline bool compare_key_bytes(
+    const char* p, const char* stored, std::size_t len) noexcept {
+    // [[maybe_unused]]: only the NEON/SSE2/LSX branches read these; the scalar
+    // fallback build (no SIMD) leaves them unused, which is an error under -Werror.
+    [[maybe_unused]] alignas(16) static constexpr uint8_t idx16[16] =
+        {0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15};
+    if constexpr (MaxKeyLen <= 16) {
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+        uint8x16_t vp = vld1q_u8(reinterpret_cast<const uint8_t*>(p));
+        uint8x16_t vs = vld1q_u8(reinterpret_cast<const uint8_t*>(stored));
+        uint8x16_t mask = vcltq_u8(vld1q_u8(idx16), vdupq_n_u8(static_cast<uint8_t>(len)));
+        uint8x16_t diff = veorq_u8(vandq_u8(vp, mask), vs);
+        return neon_all_bytes_zero(diff);
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+        __m128i vp = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p));
+        __m128i vs = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored));
+        __m128i idx = _mm_load_si128(reinterpret_cast<const __m128i*>(idx16));
+        __m128i mask = _mm_cmplt_epi8(idx, _mm_set1_epi8(static_cast<char>(len)));
+        __m128i eq = _mm_cmpeq_epi8(_mm_and_si128(vp, mask), vs);
+        return _mm_movemask_epi8(eq) == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+        __m128i vp = __lsx_vld(reinterpret_cast<const void*>(p), 0);
+        __m128i vs = __lsx_vld(reinterpret_cast<const void*>(stored), 0);
+        __m128i idx = __lsx_vld(reinterpret_cast<const void*>(idx16), 0);
+        __m128i mask = __lsx_vslt_b(idx, __lsx_vreplgr2vr_b(static_cast<int>(len)));
+        __m128i eq = __lsx_vseq_b(__lsx_vand_v(vp, mask), vs);
+        return (static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq), 0)) & 0xFFFFu) == 0xFFFFu;
+#else
+        for (std::size_t i = 0; i < len; ++i)
+            if (p[i] != stored[i]) return false;
+        return true;
+#endif
+    } else if constexpr (MaxKeyLen <= 32) {
+        [[maybe_unused]] alignas(16) static constexpr uint8_t idx32_hi[16] =
+            {16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31};
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+        uint8x16_t vp_lo = vld1q_u8(reinterpret_cast<const uint8_t*>(p));
+        uint8x16_t vp_hi = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + 16);
+        uint8x16_t vs_lo = vld1q_u8(reinterpret_cast<const uint8_t*>(stored));
+        uint8x16_t vs_hi = vld1q_u8(reinterpret_cast<const uint8_t*>(stored) + 16);
+        uint8x16_t lenv = vdupq_n_u8(static_cast<uint8_t>(len));
+        uint8x16_t m_lo = vcltq_u8(vld1q_u8(idx16),    lenv);
+        uint8x16_t m_hi = vcltq_u8(vld1q_u8(idx32_hi), lenv);
+        uint8x16_t d_lo = veorq_u8(vandq_u8(vp_lo, m_lo), vs_lo);
+        uint8x16_t d_hi = veorq_u8(vandq_u8(vp_hi, m_hi), vs_hi);
+        return neon_all_bytes_zero(vorrq_u8(d_lo, d_hi));
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+        __m128i vp_lo = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p));
+        __m128i vp_hi = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + 16));
+        __m128i vs_lo = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored));
+        __m128i vs_hi = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored + 16));
+        __m128i lenv = _mm_set1_epi8(static_cast<char>(len));
+        __m128i m_lo = _mm_cmplt_epi8(_mm_load_si128(reinterpret_cast<const __m128i*>(idx16)),    lenv);
+        __m128i m_hi = _mm_cmplt_epi8(_mm_load_si128(reinterpret_cast<const __m128i*>(idx32_hi)), lenv);
+        __m128i eq_lo = _mm_cmpeq_epi8(_mm_and_si128(vp_lo, m_lo), vs_lo);
+        __m128i eq_hi = _mm_cmpeq_epi8(_mm_and_si128(vp_hi, m_hi), vs_hi);
+        return (_mm_movemask_epi8(eq_lo) & _mm_movemask_epi8(eq_hi)) == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+        __m128i lenv = __lsx_vreplgr2vr_b(static_cast<int>(len));
+        __m128i vp_lo = __lsx_vld(reinterpret_cast<const void*>(p), 0);
+        __m128i vp_hi = __lsx_vld(reinterpret_cast<const void*>(p + 16), 0);
+        __m128i vs_lo = __lsx_vld(reinterpret_cast<const void*>(stored), 0);
+        __m128i vs_hi = __lsx_vld(reinterpret_cast<const void*>(stored + 16), 0);
+        __m128i m_lo = __lsx_vslt_b(__lsx_vld(reinterpret_cast<const void*>(idx16), 0),    lenv);
+        __m128i m_hi = __lsx_vslt_b(__lsx_vld(reinterpret_cast<const void*>(idx32_hi), 0), lenv);
+        __m128i eq_lo = __lsx_vseq_b(__lsx_vand_v(vp_lo, m_lo), vs_lo);
+        __m128i eq_hi = __lsx_vseq_b(__lsx_vand_v(vp_hi, m_hi), vs_hi);
+        unsigned mlo = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq_lo), 0)) & 0xFFFFu;
+        unsigned mhi = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq_hi), 0)) & 0xFFFFu;
+        return (mlo & mhi) == 0xFFFFu;
+#else
+        for (std::size_t i = 0; i < len; ++i)
+            if (p[i] != stored[i]) return false;
+        return true;
+#endif
+    } else if constexpr (MaxKeyLen <= 64) {
+        // 3 or 4 16-byte blocks (33..48 -> 3, 49..64 -> 4). stored is zero-padded
+        // to exactly num_blocks*16 bytes (KEY_STRIDE), so neither load overruns it.
+        // [[maybe_unused]]: the scalar fallback (no NEON/SSE2/LSX) does not use it.
+        [[maybe_unused]] constexpr std::size_t num_blocks = (MaxKeyLen + 15) / 16;
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+        uint8x16_t base = vld1q_u8(idx16);
+        uint8x16_t lenv = vdupq_n_u8(static_cast<uint8_t>(len));
+        uint8x16_t acc  = vdupq_n_u8(0);
+        for (std::size_t b = 0; b < num_blocks; ++b) {
+            uint8x16_t vp   = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + b * 16);
+            uint8x16_t vs   = vld1q_u8(reinterpret_cast<const uint8_t*>(stored) + b * 16);
+            uint8x16_t idxv = vaddq_u8(base, vdupq_n_u8(static_cast<uint8_t>(b * 16)));
+            uint8x16_t mask = vcltq_u8(idxv, lenv);
+            acc = vorrq_u8(acc, veorq_u8(vandq_u8(vp, mask), vs));
+        }
+        return neon_all_bytes_zero(acc);
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+        __m128i base = _mm_load_si128(reinterpret_cast<const __m128i*>(idx16));
+        __m128i lenv = _mm_set1_epi8(static_cast<char>(len));
+        int eq = 0xFFFF;
+        for (std::size_t b = 0; b < num_blocks; ++b) {
+            __m128i vp   = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + b * 16));
+            __m128i vs   = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored + b * 16));
+            __m128i idxv = _mm_add_epi8(base, _mm_set1_epi8(static_cast<char>(b * 16)));
+            __m128i mask = _mm_cmplt_epi8(idxv, lenv);
+            eq &= _mm_movemask_epi8(_mm_cmpeq_epi8(_mm_and_si128(vp, mask), vs));
+        }
+        return eq == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+        __m128i base = __lsx_vld(reinterpret_cast<const void*>(idx16), 0);
+        __m128i lenv = __lsx_vreplgr2vr_b(static_cast<int>(len));
+        unsigned acc = 0xFFFFu;
+        for (std::size_t b = 0; b < num_blocks; ++b) {
+            __m128i vp   = __lsx_vld(reinterpret_cast<const void*>(p + b * 16), 0);
+            __m128i vs   = __lsx_vld(reinterpret_cast<const void*>(stored + b * 16), 0);
+            __m128i idxv = __lsx_vadd_b(base, __lsx_vreplgr2vr_b(static_cast<int>(b * 16)));
+            __m128i mask = __lsx_vslt_b(idxv, lenv);
+            __m128i eq   = __lsx_vseq_b(__lsx_vand_v(vp, mask), vs);
+            acc &= static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq), 0)) & 0xFFFFu;
+        }
+        return acc == 0xFFFFu;
+#else
+        for (std::size_t i = 0; i < len; ++i)
+            if (p[i] != stored[i]) return false;
+        return true;
+#endif
+    } else {
+        for (std::size_t i = 0; i < len; ++i)
+            if (p[i] != stored[i]) return false;
+        return true;
+    }
+}
+
+// --- Single 8-bit window fast path ------------------------------------------
+//
+// Many small key sets can be told apart by inspecting a *single* 8-bit window of
+// the key bytes -- and that window need not be byte-aligned. Because every JSON
+// key is terminated by a '"', the bytes at and before a key's length are well
+// defined for any key at least that long: byte i is the key character when i is
+// inside the key and the closing quote when i == len. So we read two bytes at a
+// fixed offset, extract 8 consecutive bits at a fixed intra-byte shift, and if
+// that value is distinct for every key, a 256-entry table maps it straight to a
+// candidate key. The match then needs no hash and -- in the length-free overload
+// -- no SIMD length scan: load two bytes, shift, mask, index the table, and
+// confirm the candidate with one comparison.
+//
+// Allowing an *unaligned* window (a shift of 1..7) mixes bits from two adjacent
+// bytes and discriminates key sets that no single aligned byte can. For example,
+// the partial_tweets keys {created_at,id,text,in_reply_to_status_id,user,
+// retweet_count,favorite_count} share a colliding byte at every aligned position
+// 0,1,2, yet the 8 bits starting at bit offset 2 are unique across all seven.
+// Simpler cases fall out as the shift==0 special case: {"id","screen_name"}
+// splits on byte 0, {"jo","joe"} on the quote at byte 2.
+//
+// The window is confined to the first (shortest key length + 1) bytes so the
+// two-byte read never crosses a key's closing quote into uncontrolled value
+// bytes; that final byte is the shortest key's quote.
+template <std::size_t N, std::size_t MaxKeyLen>
+struct window_data {
+    static constexpr std::size_t KEY_STRIDE = ((MaxKeyLen + 15) / 16) * 16;
+    bool                                            ok{false};
+    std::uint8_t                                    byte_offset{0}; // first byte of the 2-byte read
+    std::uint8_t                                    shift{0};       // intra-byte bit shift (0..7)
+    std::array<std::uint8_t, 256>                   window_to_key{}; // window byte -> key index, N if none
+    std::array<std::uint8_t, N>                     key_len{};
+    std::array<std::array<char, KEY_STRIDE>, N>     key_bytes{};     // zero-padded
+};
+
+// Byte seen at `idx` for key `i`: a key character when idx is inside the key, or
+// the closing '"' when idx == len (callers keep idx <= every key's length).
+template <std::size_t N>
+constexpr unsigned window_byte_at(const std::array<std::string_view, N>& keys,
+                                  std::size_t i, std::size_t idx) noexcept {
+    if (idx < keys[i].size()) { return static_cast<unsigned char>(keys[i][idx]); }
+    return static_cast<unsigned>('"');
+}
+
+// The 8-bit window value for key `i` at (byte_offset, shift): the little-endian
+// pair (byte[off], byte[off+1]) shifted right by `shift` and truncated. Matches
+// the runtime read exactly.
+template <std::size_t N>
+constexpr unsigned window_value(const std::array<std::string_view, N>& keys,
+                                std::size_t i, std::size_t byte_offset, std::size_t shift) noexcept {
+    unsigned lo = window_byte_at<N>(keys, i, byte_offset);
+    unsigned hi = window_byte_at<N>(keys, i, byte_offset + 1);
+    return ((lo | (hi << 8)) >> shift) & 0xFFu;
+}
+
+template <std::size_t N, std::size_t MaxKeyLen>
+consteval window_data<N, MaxKeyLen>
+compute_window(const std::array<std::string_view, N>& keys) {
+    window_data<N, MaxKeyLen> out{};
+
+    std::size_t min_len = keys[0].size();
+    for (std::size_t i = 1; i < N; ++i) {
+        if (keys[i].size() < min_len) { min_len = keys[i].size(); }
+    }
+
+    // Iterate windows nearest the front first (cheapest to read, smallest shift).
+    for (std::size_t off = 0; off <= min_len; ++off) {
+        for (std::size_t shift = 0; shift < 8; ++shift) {
+            // The read touches byte off, and byte off+1 when shift != 0. Both must
+            // stay within the safe region [0, min_len] (min_len is the shortest
+            // key's quote index). off <= min_len is guaranteed by the loop bound.
+            if (shift != 0 && off + 1 > min_len) { continue; }
+
+            bool distinct = true;
+            for (std::size_t i = 0; i < N && distinct; ++i) {
+                for (std::size_t j = i + 1; j < N; ++j) {
+                    if (window_value<N>(keys, i, off, shift) == window_value<N>(keys, j, off, shift)) {
+                        distinct = false;
+                        break;
+                    }
+                }
+            }
+            if (!distinct) { continue; }
+
+            out.ok          = true;
+            out.byte_offset = static_cast<std::uint8_t>(off);
+            out.shift       = static_cast<std::uint8_t>(shift);
+            for (std::size_t b = 0; b < 256; ++b) { out.window_to_key[b] = static_cast<std::uint8_t>(N); }
+            for (std::size_t i = 0; i < N; ++i) {
+                out.window_to_key[window_value<N>(keys, i, off, shift)] = static_cast<std::uint8_t>(i);
+                out.key_len[i] = static_cast<std::uint8_t>(keys[i].size());
+                for (std::size_t c = 0; c < keys[i].size(); ++c) { out.key_bytes[i][c] = keys[i][c]; }
+            }
+            return out;
+        }
+    }
+    return out; // ok == false: no single 8-bit window distinguishes the keys
+}
+
+// Read the 8-bit window at (byte_offset, shift) from a padded key pointer.
+simdjson_really_inline std::uint8_t read_window(const char* p, std::size_t byte_offset,
+                                                std::size_t shift) noexcept {
+    std::uint16_t w;
+    // Two controlled bytes (within the shortest key + its quote, hence within the
+    // padded buffer). memcpy is the portable little-endian unaligned load.
+    std::memcpy(&w, p + byte_offset, sizeof(w));
+#if defined(__BYTE_ORDER__) && (__BYTE_ORDER__ == __ORDER_BIG_ENDIAN__)
+    w = static_cast<std::uint16_t>((w >> 8) | (w << 8));
+#endif
+    return static_cast<std::uint8_t>((w >> shift) & 0xFFu);
+}
+
+// Verify a window candidate. The window table already mapped the key to the
+// single possible index `ki` (< N); here we confirm it. Folding over 0..N-1
+// turns the runtime `ki` into a compile-time index in the matching arm, so the
+// candidate's length and bytes are constants for compare_key_bytes -- the same
+// specialization the ordered find_field path gets from a CT-length
+// unsafe_is_equal, and what keeps this path competitive. The closing-quote check
+// (p[L] == '"') both confirms the key ends exactly at the candidate's length and
+// rejects a wrong-length key, so this works whether or not the caller knew len.
+template <std::size_t N, std::size_t MaxKeyLen, std::size_t... Is>
+simdjson_really_inline std::size_t
+match_window_candidate(const char* p, std::uint8_t ki,
+                       const window_data<N, MaxKeyLen>& w,
+                       std::index_sequence<Is...>) noexcept {
+  std::size_t result = N;
+  auto try_match = [&](auto Ic) {
+    constexpr std::size_t i = decltype(Ic)::value;
+    if (ki == i && p[w.key_len[i]] == '"' &&
+        key_selector_detail::compare_key_bytes<MaxKeyLen>(
+            p, w.key_bytes[i].data(), w.key_len[i])) {
+      result = i;
+    }
+  };
+  (try_match(std::integral_constant<std::size_t, Is>{}), ...);
+  return result;
+}
+
+// --- describe() string helpers (constexpr; no std::to_string, which is not) ---
+
+// Append the decimal form of v to s.
+SIMDJSON_CONSTEXPR_STRING void append_uint(std::string& s, std::size_t v) {
+    if (v == 0) { s.push_back('0'); return; }
+    char buf[20];
+    std::size_t n = 0;
+    while (v > 0) { buf[n++] = static_cast<char>('0' + (v % 10)); v /= 10; }
+    while (n > 0) { s.push_back(buf[--n]); }
+}
+
+// Append a byte as its decimal value, plus the printable character in quotes
+// when it is in the printable ASCII range (e.g. "110 ('n')").
+SIMDJSON_CONSTEXPR_STRING void append_byte(std::string& s, unsigned b) {
+    append_uint(s, b);
+    if (b >= 0x20 && b < 0x7f) {
+        s += " ('";
+        s.push_back(static_cast<char>(b));
+        s += "')";
+    }
+}
+
+} // namespace key_selector_detail
+
+/**
+ * Stateless, compile-time key selector.
+ *
+ * Usage:
+ *   using sel_t = key_selector<"id", "text", "user">;
+ *   std::size_t i = sel_t::match_raw(raw_key); // returns sel_t::size() on miss
+ *
+ * The perfect hash is built at compile time (gperf-style, with a
+ * Hash-and-Displace fallback) and only flat tables survive to runtime. All
+ * tables are static constexpr, so the lookup fully inlines.
+ *
+ * Limitations:
+ *   - Each key must be at most 63 characters long (and no longer than
+ *     SIMDJSON_PADDING). Longer keys trigger a compile-time error.
+ *   - The number of keys should be moderate. The hard limit is 255 keys;
+ *     compilation time grows with the number of keys, so prefer a few dozen at
+ *     most per selector.
+ *   - Keys must be distinct, non-empty, and free of backslash, double-quote and
+ *     null bytes (matching is done against the raw, unescaped JSON key bytes).
+ */
+template <constevalutil::fixed_string... Keys>
+struct key_selector {
+    static constexpr std::size_t N = sizeof...(Keys);
+    static_assert(N > 0,   "key_selector requires at least one key");
+    static_assert(N <= 255,"key_selector supports at most 255 keys");
+
+    static constexpr std::array<std::string_view, N> keys{ Keys.view()... };
+    static constexpr std::size_t max_key_len = key_selector_detail::compute_max_key_len<N>(keys);
+    static_assert(max_key_len <= SIMDJSON_PADDING,
+                  "key longer than SIMDJSON_PADDING is not supported");
+    // The SIMD key-length scan covers offsets 0..63 (four 16-byte blocks), which
+    // stays within the 64-byte padding guarantee. A 64-character key's closing
+    // quote would land at offset 64 and be missed on NEON/SSE2/LSX while still
+    // matching in scalar builds, so cap at 63 to keep implementations in agreement.
+    static_assert(max_key_len <= 63,
+                  "key_selector keys must be at most 63 characters long");
+
+    static constexpr auto result = key_selector_detail::compute_phf<N>(keys);
+    static constexpr std::size_t table_size = result.table_size;
+
+    static constexpr auto phf =
+        key_selector_detail::build_phf_data<N, table_size, max_key_len>(keys, result);
+
+    // Single 8-bit-window discriminator (when one exists). Detected at compile
+    // time and selected with `if constexpr` below, so the hash path is compiled
+    // out for key sets that qualify, and this is compiled out for those that do
+    // not.
+    static constexpr auto window =
+        key_selector_detail::compute_window<N, max_key_len>(keys);
+
+    static constexpr std::size_t size() noexcept { return N; }
+
+    /**
+     * Look up a JSON key whose length is already known. p must point at the first
+     * key byte (just after the opening quote) in a padded simdjson buffer, and len
+     * must be the number of raw key bytes (the distance to the closing quote).
+     * Returns the selector index in [0, N) on match, or N on miss.
+     *
+     * Prefer this overload when the caller can obtain the key length cheaply (for
+     * example, object::for_each derives it from the structural index rather than
+     * re-scanning for the closing quote).
+     */
+    static simdjson_really_inline std::size_t match_raw(const char* p, std::size_t len) noexcept {
+        if (len == 0 || len > max_key_len) { return N; }
+
+        if constexpr (window.ok) {
+            // One 8-bit window selects the only possible candidate key;
+            // match_window_candidate confirms it (bytes + closing quote). p sits
+            // in a padded buffer and the window stays within the shortest key +
+            // quote, so the two-byte read is always in bounds. len is unused here
+            // because the quote check already pins the key's end.
+            std::uint8_t ki = window.window_to_key[
+                key_selector_detail::read_window(p, window.byte_offset, window.shift)];
+            if (ki >= N) { return N; }
+            return key_selector_detail::match_window_candidate(
+                p, ki, window, std::make_index_sequence<N>{});
+        }
+
+        std::size_t slot;
+        if (phf.num_positions == key_selector_detail::HD_MODE) {
+            // Hash-and-Displace: bucket displacement + per-key hash.
+            std::string_view key(p, len);
+            std::size_t bucket = key_selector_detail::hd_bucket_hash(key);
+            std::size_t kh = (phf.hd_hash_variant == key_selector_detail::HD_HASH_2BYTE_FLAG)
+                ? key_selector_detail::hd_key_hash_2(key)
+                : key_selector_detail::hd_key_hash_4(key);
+            slot = (phf.asso_values[0][bucket] + kh) & (table_size - 1);
+        } else {
+            // gperf: h = len + sum of asso_values over the selected positions.
+            // positions / num_positions / asso_values are compile-time constants,
+            // so this loop fully unrolls. The idx < len guard mirrors the
+            // generator's char_at()-> 256 -> skip behavior for out-of-range
+            // positions (required: arbitrary positions may exceed a key's length).
+            std::size_t h = len;
+            for (std::uint8_t i = 0; i < phf.num_positions; ++i) {
+                std::uint8_t pos = phf.positions[i];
+                std::size_t idx = (pos == key_selector_detail::POS_LAST_CHAR)
+                                  ? (len - std::size_t{1})
+                                  : static_cast<std::size_t>(pos);
+                if (idx < len) {
+                    h += phf.asso_values[i][static_cast<unsigned char>(p[idx])];
+                }
+            }
+            slot = h & (table_size - 1);
+        }
+
+        std::uint8_t ki = phf.slot_to_key[slot];
+        if (ki >= N) { return N; }
+        if (phf.slot_key_len[slot] != len) { return N; }
+        if (!key_selector_detail::compare_key_bytes<max_key_len>(
+                p, phf.slot_key_bytes[slot].data(), len)) { return N; }
+        return ki;
+    }
+
+    /**
+     * Look up a JSON key. rjs must point just after an opening quote in a padded
+     * simdjson buffer. Returns the selector index in [0, N) on match, or N on miss.
+     * The key length is recovered with a SIMD scan for the closing quote; callers
+     * that already know the length should use the (p, len) overload above.
+     */
+    static simdjson_really_inline std::size_t match_raw(raw_json_string rjs) noexcept {
+        const char* p = rjs.raw();
+        if constexpr (window.ok) {
+            // One 8-bit window picks the candidate; verifying the candidate's
+            // bytes and its closing '"' confirms the full key, so the length scan
+            // is unnecessary. The window read is in bounds (padding), and the
+            // candidate length is at most max_key_len.
+            std::uint8_t ki = window.window_to_key[
+                key_selector_detail::read_window(p, window.byte_offset, window.shift)];
+            if (ki >= N) { return N; }
+            return key_selector_detail::match_window_candidate(
+                p, ki, window, std::make_index_sequence<N>{});
+        }
+        return match_raw(p, key_selector_detail::scan_key_length<max_key_len>(p));
+    }
+
+    /** Return the key text at selector index i (i in [0, N)). */
+    static constexpr std::string_view key_at(std::size_t i) noexcept {
+        return keys[i];
+    }
+
+    /**
+     * Return a complete, human-readable, multi-line description of how this
+     * selector classifies a key: which algorithm was selected at compile time
+     * (single 8-bit window, gperf-style perfect hash, or hash-and-displace), the
+     * exact bytes/positions it inspects, and the contents of the lookup tables
+     * (which window bytes or hash slots map to which key). The text mirrors what
+     * match_raw() does step by step.
+     *
+     * Everything it reports is derived from the compile-time tables, so describe()
+     * is itself usable in a constant expression when the standard library supports
+     * constexpr std::string (__cpp_lib_constexpr_string):
+     *
+     *   static_assert(!key_selector<"name", "city">::describe().empty());
+     *
+     * It allocates a std::string and is meant for documentation, debugging and
+     * tests, not for any hot path.
+     */
+    static SIMDJSON_CONSTEXPR_STRING std::string describe() {
+        std::string s;
+        s += "key_selector: ";
+        key_selector_detail::append_uint(s, N);
+        s += " keys, max key length ";
+        key_selector_detail::append_uint(s, max_key_len);
+        s += "\nkeys:\n";
+        for (std::size_t i = 0; i < N; ++i) {
+            s += "  [";
+            key_selector_detail::append_uint(s, i);
+            s += "] \"";
+            s += keys[i];
+            s += "\" (length ";
+            key_selector_detail::append_uint(s, keys[i].size());
+            s += ")\n";
+        }
+        if constexpr (window.ok) {
+            // Mirrors the window fast path of match_raw().
+            s += "algorithm: single 8-bit window\n";
+            s += "  step 1: read 2 bytes at offset ";
+            key_selector_detail::append_uint(s, window.byte_offset);
+            s += ", interpret them as a little-endian 16-bit value, shift right by ";
+            key_selector_detail::append_uint(s, window.shift);
+            s += " bits, and keep the low 8 bits\n";
+            s += "  step 2: map that byte through a 256-entry table to a key index (";
+            key_selector_detail::append_uint(s, N);
+            s += " means no match):\n";
+            for (std::size_t b = 0; b < 256; ++b) {
+                if (window.window_to_key[b] < N) {
+                    s += "    byte ";
+                    key_selector_detail::append_byte(s, static_cast<unsigned>(b));
+                    s += " -> key ";
+                    key_selector_detail::append_uint(s, window.window_to_key[b]);
+                    s += "\n";
+                }
+            }
+            s += "  step 3: confirm the candidate by checking the closing quote sits at the key's length and comparing the key bytes\n";
+        } else {
+            // Mirrors the perfect-hash path of match_raw().
+            if constexpr (phf.num_positions == key_selector_detail::HD_MODE) {
+                s += "algorithm: hash-and-displace perfect hash\n";
+                s += "  step 1: bucket = (first_byte + last_byte*3 + length*17) mod 256\n";
+                s += "  step 2: keyhash = base-31 rolling hash of the length and the first ";
+                key_selector_detail::append_uint(s, phf.hd_hash_variant);
+                s += " bytes\n";
+                s += "  step 3: slot = (displacement[bucket] + keyhash) mod ";
+                key_selector_detail::append_uint(s, table_size);
+                s += "\n  step 4: slot_to_key[slot] gives the candidate key index\n";
+                s += "  per-key derivation:\n";
+                for (std::size_t i = 0; i < N; ++i) {
+                    std::string_view k = keys[i];
+                    std::size_t bucket = key_selector_detail::hd_bucket_hash(k);
+                    std::size_t kh = (phf.hd_hash_variant == key_selector_detail::HD_HASH_2BYTE_FLAG)
+                        ? key_selector_detail::hd_key_hash_2(k)
+                        : key_selector_detail::hd_key_hash_4(k);
+                    std::size_t slot = (phf.asso_values[0][bucket] + kh) & (table_size - 1);
+                    s += "    \"";
+                    s += k;
+                    s += "\": bucket=";
+                    key_selector_detail::append_uint(s, bucket);
+                    s += " displacement=";
+                    key_selector_detail::append_uint(s, phf.asso_values[0][bucket]);
+                    s += " keyhash=";
+                    key_selector_detail::append_uint(s, kh);
+                    s += " slot=";
+                    key_selector_detail::append_uint(s, slot);
+                    s += "\n";
+                }
+            } else {
+                s += "algorithm: gperf-style perfect hash over ";
+                key_selector_detail::append_uint(s, phf.num_positions);
+                s += " character position(s)\n";
+                s += "  step 1: h = key length\n";
+                s += "  step 2: for each position below, add its association value for the key byte there (a position past the key's length contributes 0):\n";
+                for (std::size_t i = 0; i < phf.num_positions; ++i) {
+                    s += "    position ";
+                    if (phf.positions[i] == key_selector_detail::POS_LAST_CHAR) {
+                        s += "last character";
+                    } else {
+                        s += "byte index ";
+                        key_selector_detail::append_uint(s, phf.positions[i]);
+                    }
+                    s += "\n";
+                }
+                s += "  step 3: slot = h mod ";
+                key_selector_detail::append_uint(s, table_size);
+                s += " (a power of two, applied as a bitmask)\n";
+                s += "  step 4: slot_to_key[slot] gives the candidate key index\n";
+                s += "  per-key derivation:\n";
+                for (std::size_t i = 0; i < N; ++i) {
+                    std::string_view k = keys[i];
+                    std::size_t h = k.size();
+                    for (std::size_t pi = 0; pi < phf.num_positions; ++pi) {
+                        std::size_t pos = phf.positions[pi];
+                        std::size_t idx = (pos == key_selector_detail::POS_LAST_CHAR)
+                                          ? (k.size() - 1) : pos;
+                        if (idx < k.size()) {
+                            h += phf.asso_values[pi][static_cast<unsigned char>(k[idx])];
+                        }
+                    }
+                    std::size_t slot = h & (table_size - 1);
+                    s += "    \"";
+                    s += k;
+                    s += "\": h=";
+                    key_selector_detail::append_uint(s, h);
+                    s += " slot=";
+                    key_selector_detail::append_uint(s, slot);
+                    s += "\n";
+                }
+            }
+            s += "  occupied slots (slot -> key):\n";
+            for (std::size_t slot = 0; slot < table_size; ++slot) {
+                if (phf.slot_to_key[slot] < N) {
+                    s += "    slot ";
+                    key_selector_detail::append_uint(s, slot);
+                    s += " -> key ";
+                    key_selector_detail::append_uint(s, phf.slot_to_key[slot]);
+                    s += " (\"";
+                    s += keys[phf.slot_to_key[slot]];
+                    s += "\", length ";
+                    key_selector_detail::append_uint(s, phf.slot_key_len[slot]);
+                    s += ")\n";
+                }
+            }
+            s += "  confirm the candidate by checking the key length matches and comparing the key bytes\n";
+        }
+        return s;
+    }
+};
+
+namespace key_selector_detail {
+template <typename> struct is_key_selector : std::false_type {};
+template <constevalutil::fixed_string... Keys>
+struct is_key_selector<key_selector<Keys...>> : std::true_type {};
+} // namespace key_selector_detail
+
+/**
+ * Matches any instantiation of key_selector<Keys...>. Used to constrain
+ * object::for_each so that passing a non-selector type yields a clear
+ * constraint error rather than a cascade of failures inside for_each.
+ */
+template <typename T>
+concept key_selector_type = key_selector_detail::is_key_selector<T>::value;
+
+} // namespace ondemand
+} // namespace westmere
+} // namespace simdjson
+
+#endif // SIMDJSON_SUPPORTS_CONCEPTS
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+/* end file simdjson/generic/ondemand/key_selector.h for westmere */
 /* including simdjson/generic/ondemand/object.h for westmere: #include "simdjson/generic/ondemand/object.h" */
 /* begin file simdjson/generic/ondemand/object.h for westmere */
 #ifndef SIMDJSON_GENERIC_ONDEMAND_OBJECT_H
@@ -136836,6 +174308,7 @@ public:
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/implementation_simdjson_result_base.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/key_selector.h" */
 /* amalgamation skipped (editor-only): #include <vector> */
 /* amalgamation skipped (editor-only): #if SIMDJSON_STATIC_REFLECTION && SIMDJSON_SUPPORTS_CONCEPTS */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_string_builder.h"  // for constevalutil::fixed_string */
@@ -136846,6 +174319,114 @@ namespace simdjson {
 namespace westmere {
 namespace ondemand {

+#if SIMDJSON_SUPPORTS_CONCEPTS
+/**
+ * Result of object::for_each: the first error encountered (SUCCESS if none) and
+ * the number of distinct selector keys that matched during the walk. A
+ * matched_count equal to Selector::size() means every selected key was present
+ * in the object. Implicitly converts to error_code so existing callers that only
+ * care about the error (including SIMDJSON_TRY and the test ASSERT_* macros) keep
+ * working unchanged.
+ *
+ * On error paths (a handler returns a non-SUCCESS error_code, or a direct-target
+ * deserialization via value::get fails), matched_count reflects only the number
+ * of distinct selected keys that were successfully handled *before* the failure;
+ * the failing key is not counted. This is the current behavior.
+ *
+ * Marked [[nodiscard]]: for_each surfaces parse/type errors (from a matched
+ * value, a returning handler, or the structural walk itself) only through this
+ * result, so it must not be silently dropped. This matters most in builds
+ * without exceptions, where it is the only channel for those errors. To
+ * deliberately ignore it, assign to a variable or cast to void.
+ */
+struct [[nodiscard]] for_each_result {
+  error_code error{SUCCESS};
+  std::size_t matched_count{0};
+  constexpr operator error_code() const noexcept { return error; }
+};
+
+namespace key_selector_for_each_detail {
+/**
+ * A per-key handler for the variadic object::for_each is either:
+ *   - an invocable taking a value (run custom logic for that field), or
+ *   - a deserialization target T, in which case the matched value is assigned
+ *     directly via value::get(T&).
+ * The target form lets callers bind fields straight to variables without a
+ * lambda per key -- e.g. obj.for_each<"name","city","age">(name, city, age) --
+ * while still allowing a lambda in any position when custom logic is needed
+ * (for example to descend into a nested object).
+ */
+template <typename H>
+concept field_handler =
+    std::is_invocable_v<std::remove_reference_t<H>&, value> ||
+    ::simdjson::deserializable<std::remove_cvref_t<H>, value>;
+
+/**
+ * noexcept-ness of handling a single handler: nothrow-invocability for the
+ * callback form, or nothrow-deserializability for the direct-target form
+ * (builtin scalar/string targets are noexcept; custom targets follow their
+ * own tag_invoke noexcept specification).
+ */
+template <typename H>
+inline constexpr bool nothrow_field_handler_v =
+    std::is_invocable_v<std::remove_reference_t<H>&, value>
+        ? std::is_nothrow_invocable_v<std::remove_reference_t<H>&, value>
+        : ::simdjson::nothrow_deserializable<std::remove_cvref_t<H>, value>;
+
+/**
+ * nothrow_field_handler_v folded over a whole handler pack. The variadic
+ * for_each overloads spell their conditional-noexcept specifier with this
+ * single id-expression rather than an inline fold expression. MSVC (VS18)
+ * mishandles a fold-expression that appears directly in the noexcept-specifier
+ * of an out-of-line template definition: it fails to recognize the definition's
+ * specifier as matching the in-class declaration's and reports a spurious
+ * C2382 ("redefinition; different exception specifications"). Naming the fold
+ * here keeps the specifier a plain identifier, which matches reliably.
+ */
+template <typename... Handlers>
+inline constexpr bool nothrow_field_handlers_v =
+    (nothrow_field_handler_v<Handlers> && ...);
+} // namespace key_selector_for_each_detail
+#endif
+
+/**
+ * An opaque snapshot of an object's scanning position, obtained from
+ * object::get_current_position() and consumed by object::revert_position().
+ *
+ * This bundles the raw token position with the iteration depth that was
+ * live at the moment of capture: a scalar field value (e.g. a number)
+ * unwinds back to the object's own depth once fully consumed, while a
+ * compound value (array/object) not yet fully consumed stays one level
+ * deeper. The correct depth to restore to therefore depends on what was
+ * live when the snapshot was taken, not on the object itself, which is
+ * why both pieces travel together here rather than being recomputed later.
+ *
+ * The members are private and only constructible by object itself: a
+ * hand-built value here would let revert_position() jump to an arbitrary,
+ * unvalidated position and depth, and this type carries no way to check
+ * that a given snapshot still refers to a live object. Only ever pass
+ * along a value you got from get_current_position(), and only back to the
+ * same object, before that object has been reset() or the parser has
+ * iterate()d a new document: neither invalidates a snapshot in a way this
+ * type can detect.
+ */
+struct object_position {
+  /**
+   * Default-constructed so a variable can be declared and assigned later,
+   * matching e.g. document()/object(). Not a valid position to revert to.
+   */
+  simdjson_inline object_position() noexcept = default;
+
+private:
+  token_position position{};
+  depth_t depth{};
+
+  simdjson_inline object_position(token_position position_, depth_t depth_) noexcept
+    : position(position_), depth(depth_) {}
+
+  friend class object;
+};
+
 /**
  * A forward-only JSON object field iterator.
  */
@@ -136864,8 +174445,19 @@ public:
    * Using the iterator directly is also possible but error-prone and discouraged. In particular,
    * you must dereference the iterator exactly once per iteration (before calling '++').
    * Doing otherwise is unsafe and may lead to errors. You are responsible for ensuring
+   *
+   * With SIMDJSON_DEVELOPMENT_CHECKS, the iterator locks this object while it is
+   * alive, so that reentrant access (find_field(), reset(), ...) is reported as
+   * OUT_OF_ORDER_ITERATION.
+   */
+  simdjson_inline simdjson_result<object_iterator> begin() & noexcept;
+  /**
+   * Get an iterator to the start of a temporary object, e.g., `v.get_object().begin()`.
+   *
+   * The iterator does not depend on the object instance and may outlive it, so
+   * it does not lock it.
    */
-  simdjson_inline simdjson_result<object_iterator> begin() noexcept;
+  simdjson_inline simdjson_result<object_iterator> begin() && noexcept;
   simdjson_inline simdjson_result<object_iterator> end() noexcept;
   /**
    * Look up a field by name on an object (order-sensitive). By order-sensitive, we mean that
@@ -136877,10 +174469,11 @@ public:
    *
    * ```cpp
    * simdjson::ondemand::parser parser;
-   * auto obj = parser.parse(R"( { "x": 1, "y": 2, "z": 3 } )"_padded);
-   * double z = obj.find_field("z");
-   * double y = obj.find_field("y");
-   * double x = obj.find_field("x");
+   * auto json = R"( { "x": 1, "y": 2, "z": 3 } )"_padded;
+   * auto doc = parser.iterate(json);
+   * double z = doc.find_field("z");
+   * double y = doc.find_field("y");
+   * double x = doc.find_field("x");
    * ```
    * If you have multiple fields with a matching key ({"x": 1,  "x": 1}) be mindful
    * that only one field is returned.
@@ -136953,6 +174546,100 @@ public:
   /** @overload simdjson_inline simdjson_result<value> find_field_unordered(std::string_view key) & noexcept; */
   simdjson_inline simdjson_result<value> operator[](std::string_view key) && noexcept;

+#if SIMDJSON_SUPPORTS_CONCEPTS
+  /**
+   * Walk this object once and invoke on_match(selector_index, value) for each
+   * field whose key is in the compile-time key_selector Selector, in JSON order
+   * (first occurrence of a duplicate key wins). Iteration stops once all
+   * Selector::size() keys have matched or the object ends. The value is consumed
+   * in place, so this is a low-overhead way to extract a known set of fields
+   * regardless of their order in the JSON.
+   *
+   * Like other object iteration in simdjson, for_each consumes the object by
+   * advancing the underlying iterator state; after the call the same object
+   * instance should not be used for further field access or iteration.
+   *
+   * Usage:
+   *   using sel_t = ondemand::key_selector<"id", "text", "user">;
+   *   obj.for_each<sel_t>([&](std::size_t i, ondemand::value v) {
+   *     switch (i) { case 0: ...; case 1: ...; }
+   *   });
+   *
+   * Limitations (see key_selector): each key must be at most 63 characters long,
+   * and the number of keys should be moderate (hard limit 255; a handful is
+   * best, as the compile-time perfect hash may fail or slow compilation for
+   * large key sets). The keys must be distinct, non-empty, and free of backslash, double-quote and
+   * null bytes.
+   *
+   * The callback may return either void or an error_code. When it returns an
+   * error_code, the walk stops at the first non-SUCCESS result and that error is
+   * returned, which lets the callback surface value-parse errors.
+   *
+   * This function is conditionally noexcept: it is noexcept exactly when invoking
+   * the callback is noexcept. The callback runs inside this frame, so a throwing
+   * callback (e.g. one using the exception-throwing conversions like
+   * std::string_view(value) or uint64_t(value)) makes for_each potentially
+   * throwing too -- the exception propagates to the caller instead of crossing a
+   * noexcept boundary and calling std::terminate.
+   *
+   * @returns a for_each_result holding the first error encountered while walking
+   *          the object (including any error returned by the callback, SUCCESS if
+   *          none) and the number of distinct selector keys that matched. The
+   *          result converts implicitly to error_code, so callers that only need
+   *          the error can ignore the count.
+   */
+  template <typename Selector, typename Func>
+    requires key_selector_type<Selector> &&
+             std::is_invocable_v<Func&, std::size_t, value>
+  simdjson_inline for_each_result for_each(Func&& on_match)
+      noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, value>);
+
+  /**
+   * Variadic per-key form. Provide exactly one handler per key in the Selector
+   * (compiler-enforced). Handlers are processed in JSON document order for the
+   * matching keys. Each handler is either:
+   *   - a deserialization target (a variable), in which case the matched value
+   *     is assigned to it via value::get -- no lambda required; or
+   *   - an invocable taking the ondemand::value (for custom logic such as
+   *     descending into a nested object). It may return void or error_code;
+   *     returning error_code lets you surface parse/type errors.
+   * The two styles may be mixed freely, one handler per key.
+   *
+   * Example (bind fields straight to variables):
+   *   using fields = ondemand::key_selector<"name", "city", "age">;
+   *   obj.for_each<fields>(name, city, age);
+   *
+   * Example (mixing a target and a lambda):
+   *   obj.for_each<ondemand::key_selector<"id", "user">>(
+   *     id,                                          // assigned via value::get
+   *     [&](ondemand::value v){ u = read_user(v); }  // custom logic
+   *   );
+   *
+   * The index-based single-callback form (taking (size_t, value)) remains
+   * available for shared-state or more complex per-key logic.
+   */
+  template <typename Selector, typename... Handlers>
+    requires key_selector_type<Selector> &&
+             (sizeof...(Handlers) == Selector::size()) &&
+             (key_selector_for_each_detail::field_handler<Handlers> && ...)
+  simdjson_inline for_each_result for_each(Handlers&&... on_match)
+      noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
+  /**
+   * Direct-key shorthand. Equivalent to for_each<key_selector<Keys...>>(...).
+   * Lets you write the keys inline without a separate using/alias, binding each
+   * field straight to a variable (or a lambda, see the Selector form above):
+   *
+   *   obj.for_each<"name", "city", "age">(name, city, age);
+   */
+  template <constevalutil::fixed_string... Keys, typename... Handlers>
+    requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+             (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+             (key_selector_for_each_detail::field_handler<Handlers> && ...)
+  simdjson_inline for_each_result for_each(Handlers&&... on_match)
+      noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+#endif
+
   /**
    * Get the value associated with the given JSON pointer. We use the RFC 6901
    * https://tools.ietf.org/html/rfc6901 standard, interpreting the current node
@@ -137029,6 +174716,34 @@ public:
    * @returns true if the object contains some elements (not empty)
    */
   inline simdjson_result<bool> reset() & noexcept;
+  /**
+   * Get an opaque token representing the object's current scanning position.
+   * Pass it to revert_position() to return to this exact point later, without
+   * paying the cost of a full reset() and re-scan from the beginning.
+   *
+   * A typical use is an optional field that may or may not be next: capture
+   * the position, attempt find_field(), and on NO_SUCH_FIELD, revert_position()
+   * instead of reset() so that fields already consumed are not rescanned.
+   *
+   * The returned token is only valid for this object, and only until it is
+   * reset() or the parser iterate()s a new document; using it after either
+   * is undefined behavior (see object_position).
+   *
+   * @returns An opaque position token.
+   */
+  simdjson_inline object_position get_current_position() const noexcept;
+  /**
+   * Return the object's scanning position to a snapshot previously obtained
+   * from get_current_position(). Unlike reset(), this does not rescan the
+   * object from the beginning: fields before the captured position remain
+   * consumed, and scanning resumes exactly where the snapshot was captured.
+   *
+   * @param position A snapshot previously returned by get_current_position(),
+   *        for this same object.
+   * @returns SUCCESS, or OUT_OF_ORDER_ITERATION if called during active
+   *          iteration (SIMDJSON_DEVELOPMENT_CHECKS builds only).
+   */
+  simdjson_inline error_code revert_position(object_position position) noexcept;
   /**
    * This method scans the beginning of the object and checks whether the
    * object is empty.
@@ -137074,7 +174789,7 @@ public:
    */
   template <typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out)
-     noexcept(custom_deserializable<T, object> ? nothrow_custom_deserializable<T, object> : true) {
+     noexcept(nothrow_gettable<T, object>) {
     static_assert(custom_deserializable<T, object>);
     return deserialize(*this, out);
   }
@@ -137086,7 +174801,7 @@ public:
    */
   template <typename T>
   simdjson_inline simdjson_result<T> get()
-    noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+    noexcept(nothrow_gettable<T, object>)
   {
     static_assert(std::is_default_constructible<T>::value, "The specified type is not default constructible.");
     T out{};
@@ -137138,10 +174853,18 @@ protected:
   simdjson_warn_unused simdjson_inline error_code find_field_raw(const std::string_view key) noexcept;

   value_iterator iter{};
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  bool locked{false};
+  simdjson_inline void set_locked(bool _locked) noexcept;
+#endif

   friend class value;
   friend class document;
   friend struct simdjson_result<object>;
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  friend class object_iterator;
+  friend struct simdjson_result<object_iterator>;
+#endif
 };

 } // namespace ondemand
@@ -137157,7 +174880,8 @@ public:
   simdjson_inline simdjson_result(error_code error) noexcept; ///< @private
   simdjson_inline simdjson_result() noexcept = default;

-  simdjson_inline simdjson_result<westmere::ondemand::object_iterator> begin() noexcept;
+  simdjson_inline simdjson_result<westmere::ondemand::object_iterator> begin() & noexcept;
+  simdjson_inline simdjson_result<westmere::ondemand::object_iterator> begin() && noexcept;
   simdjson_inline simdjson_result<westmere::ondemand::object_iterator> end() noexcept;
   simdjson_inline simdjson_result<westmere::ondemand::value> find_field(std::string_view key) & noexcept;
   simdjson_inline simdjson_result<westmere::ondemand::value> find_field(std::string_view key) && noexcept;
@@ -137175,6 +174899,8 @@ public:
 #endif
   simdjson_inline error_code for_each_at_path_with_wildcard(std::string_view json_path, Func&& callback) noexcept;
   inline simdjson_result<bool> reset() noexcept;
+  inline simdjson_result<westmere::ondemand::object_position> get_current_position() noexcept;
+  inline error_code revert_position(westmere::ondemand::object_position position) noexcept;
   inline simdjson_result<bool> is_empty() noexcept;
   inline simdjson_result<size_t> count_fields() & noexcept;
   inline simdjson_result<std::string_view> raw_json() noexcept;
@@ -137182,7 +174908,7 @@ public:
   // TODO: move this code into object-inl.h

   template<typename T>
-  simdjson_inline simdjson_result<T> get() noexcept {
+  simdjson_inline simdjson_result<T> get() noexcept(nothrow_gettable<T, westmere::ondemand::object>) {
     if (error()) { return error(); }
     if constexpr (std::is_same_v<T, westmere::ondemand::object>) {
       return first;
@@ -137190,7 +174916,7 @@ public:
     return first.get<T>();
   }
   template<typename T>
-  simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept {
+  simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept(nothrow_gettable<T, westmere::ondemand::object>) {
     if (error()) { return error(); }
     if constexpr (std::is_same_v<T, westmere::ondemand::object>) {
       out = first;
@@ -137200,6 +174926,39 @@ public:
     return SUCCESS;
   }

+  /**
+   * Forwards to object::for_each on the underlying object, so error-code-style
+   * chains (e.g. doc["x"].get_object()) can call for_each without first
+   * extracting the object. If this result holds an error, that error is returned
+   * (with a zero match count) and the callback is not invoked. See
+   * object::for_each for the semantics.
+   */
+  template <typename Selector, typename Func>
+    requires westmere::ondemand::key_selector_type<Selector> &&
+             std::is_invocable_v<Func&, std::size_t, westmere::ondemand::value>
+  simdjson_inline westmere::ondemand::for_each_result for_each(Func&& on_match)
+      noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, westmere::ondemand::value>);
+
+  /**
+   * Forwarding overload for the variadic per-key form (targets and/or lambdas).
+   */
+  template <typename Selector, typename... Handlers>
+    requires westmere::ondemand::key_selector_type<Selector> &&
+             (sizeof...(Handlers) == Selector::size()) &&
+             (westmere::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+  simdjson_inline westmere::ondemand::for_each_result for_each(Handlers&&... on_match)
+      noexcept(westmere::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
+  /**
+   * Forwarding overload for the direct-key variadic form.
+   */
+  template <constevalutil::fixed_string... Keys, typename... Handlers>
+    requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+             (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+             (westmere::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+  simdjson_inline westmere::ondemand::for_each_result for_each(Handlers&&... on_match)
+      noexcept(westmere::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
 #if SIMDJSON_STATIC_REFLECTION
   // TODO: move this code into object-inl.h
   template<constevalutil::fixed_string... FieldNames, typename T>
@@ -137240,6 +174999,15 @@ public:
    */
   simdjson_inline object_iterator() noexcept = default;

+#if SIMDJSON_DEVELOPMENT_CHECKS
+   simdjson_inline ~object_iterator() noexcept;
+
+   simdjson_inline object_iterator(object_iterator&&) noexcept;
+   simdjson_inline object_iterator& operator=(object_iterator&&) noexcept;
+   simdjson_inline object_iterator(const object_iterator&) noexcept;
+   simdjson_inline object_iterator& operator=(const object_iterator&) noexcept;
+#endif
+
   //
   // Iterator interface
   //
@@ -137259,6 +175027,9 @@ public:
 private:
 #if SIMDJSON_DEVELOPMENT_CHECKS
    bool has_been_referenced{false};
+   object* parent{nullptr};
+
+   simdjson_inline object_iterator(const value_iterator &_iter, object* _parent) noexcept;
 #endif
   /**
    * The underlying JSON iterator.
@@ -137304,6 +175075,191 @@ public:

 #endif // SIMDJSON_GENERIC_ONDEMAND_OBJECT_ITERATOR_H
 /* end file simdjson/generic/ondemand/object_iterator.h for westmere */
+/* including simdjson/generic/ondemand/ranges.h for westmere: #include "simdjson/generic/ondemand/ranges.h" */
+/* begin file simdjson/generic/ondemand/ranges.h for westmere */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_RANGES_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_ONDEMAND_RANGES_H */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/field.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#if SIMDJSON_SUPPORTS_RANGES
+
+namespace simdjson {
+namespace westmere {
+namespace ondemand {
+
+/**
+ * A ranges-compatible iterator adapter for JSON arrays.
+ *
+ * Wraps array_iterator to satisfy std::input_iterator by providing:
+ * - const operator* (via mutable internal state)
+ * - post-increment operator
+ * - iterator_concept tag
+ *
+ * The mutable approach is standard for single-pass input iterators that
+ * read from external sources (similar to std::istream_iterator).
+ */
+class array_range_iterator {
+public:
+  using iterator_concept = std::input_iterator_tag;
+  using value_type = simdjson_result<value>;
+  using reference = simdjson_result<value>;
+  using difference_type = std::ptrdiff_t;
+
+  simdjson_inline array_range_iterator() noexcept = default;
+  simdjson_inline explicit array_range_iterator(array_iterator iter) noexcept;
+
+  /**
+   * Get the current element. Const-qualified for std::indirectly_readable;
+   * internally delegates to the mutable wrapped iterator.
+   */
+  simdjson_inline simdjson_result<value> operator*() const noexcept;
+  simdjson_inline array_range_iterator& operator++() noexcept;
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+  simdjson_inline void operator++(int) noexcept;
+SIMDJSON_POP_DISABLE_WARNINGS
+  /**
+   * Comparison delegates to array_iterator::operator==, which checks
+   * whether the underlying parser has finished the array (depth-based).
+   */
+  simdjson_inline friend bool operator==(const array_range_iterator& a,
+                                         const array_range_iterator& b) noexcept {
+    return a.iter_ == b.iter_;
+  }
+
+private:
+  mutable array_iterator iter_{};
+};
+
+/**
+ * A std::ranges::view over a JSON array.
+ *
+ * Wraps an ondemand::array and exposes begin()/end() that return
+ * array_range_iterator (satisfying std::input_iterator), enabling
+ * use with std::views::transform and other range adaptors.
+ *
+ * If the array's begin() returns an error (only possible under
+ * SIMDJSON_DEVELOPMENT_CHECKS), the range will be empty and error()
+ * will return the error code.
+ *
+ * Usage:
+ *   ondemand::parser parser;
+ *   auto doc = parser.iterate(json);
+ *   auto arr = doc.get_array().value();
+ *   for (auto elem : ondemand::get_range(arr)) { ... }
+ */
+class array_range {
+public:
+  simdjson_inline array_range() noexcept = default;
+  simdjson_inline explicit array_range(array& arr) noexcept;
+
+  simdjson_inline array_range_iterator begin() noexcept;
+  simdjson_inline array_range_iterator end() noexcept;
+
+  /** Returns SUCCESS if the range was created successfully, or the error code otherwise. */
+  simdjson_inline error_code error() const noexcept { return error_; }
+
+private:
+  array_iterator begin_{};
+  array_iterator end_{};
+  error_code error_{SUCCESS};
+};
+
+/**
+ * A ranges-compatible iterator adapter for JSON objects.
+ *
+ * Wraps object_iterator to satisfy std::input_iterator, yielding
+ * simdjson_result<field> elements (key-value pairs).
+ */
+class object_range_iterator {
+public:
+  using iterator_concept = std::input_iterator_tag;
+  using value_type = simdjson_result<field>;
+  using reference = simdjson_result<field>;
+  using difference_type = std::ptrdiff_t;
+
+  simdjson_inline object_range_iterator() noexcept = default;
+  simdjson_inline explicit object_range_iterator(object_iterator iter) noexcept;
+
+  simdjson_inline simdjson_result<field> operator*() const noexcept;
+  simdjson_inline object_range_iterator& operator++() noexcept;
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+  simdjson_inline void operator++(int) noexcept;
+SIMDJSON_POP_DISABLE_WARNINGS
+  simdjson_inline friend bool operator==(const object_range_iterator& a,
+                                         const object_range_iterator& b) noexcept {
+    return a.iter_ == b.iter_;
+  }
+
+private:
+  mutable object_iterator iter_{};
+};
+
+/**
+ * A std::ranges::view over a JSON object.
+ *
+ * Wraps an ondemand::object and exposes begin()/end() that return
+ * object_range_iterator, enabling use with range adaptors.
+ *
+ * If the object's begin() returns an error, the range will be empty
+ * and error() will return the error code.
+ */
+class object_range {
+public:
+  simdjson_inline object_range() noexcept = default;
+  simdjson_inline explicit object_range(object& obj) noexcept;
+
+  simdjson_inline object_range_iterator begin() noexcept;
+  simdjson_inline object_range_iterator end() noexcept;
+
+  /** Returns SUCCESS if the range was created successfully, or the error code otherwise. */
+  simdjson_inline error_code error() const noexcept { return error_; }
+
+private:
+  object_iterator begin_{};
+  object_iterator end_{};
+  error_code error_{SUCCESS};
+};
+
+/** Get a std::ranges compatible view over a JSON array. */
+simdjson_inline array_range get_range(array& arr) noexcept;
+
+/** Get a std::ranges compatible view over a JSON object (key-value pairs). */
+simdjson_inline object_range get_key_value_range(object& obj) noexcept;
+
+#if SIMDJSON_EXCEPTIONS
+/** Get a std::ranges compatible view, unwrapping the simdjson_result (throws on error). */
+simdjson_inline array_range get_range(simdjson_result<array> result);
+
+/** Get a std::ranges compatible view, unwrapping the simdjson_result (throws on error). */
+simdjson_inline object_range get_key_value_range(simdjson_result<object> result);
+#endif // SIMDJSON_EXCEPTIONS
+
+} // namespace ondemand
+} // namespace westmere
+} // namespace simdjson
+
+namespace std {
+namespace ranges {
+template<>
+inline constexpr bool enable_view<simdjson::westmere::ondemand::array_range> = true;
+template<>
+inline constexpr bool enable_view<simdjson::westmere::ondemand::object_range> = true;
+} // namespace ranges
+} // namespace std
+
+#endif // SIMDJSON_SUPPORTS_RANGES
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_RANGES_H
+/* end file simdjson/generic/ondemand/ranges.h for westmere */
 /* including simdjson/generic/ondemand/serialization.h for westmere: #include "simdjson/generic/ondemand/serialization.h" */
 /* begin file simdjson/generic/ondemand/serialization.h for westmere */
 #ifndef SIMDJSON_GENERIC_ONDEMAND_SERIALIZATION_H
@@ -137436,12 +175392,14 @@ inline std::ostream& operator<<(std::ostream& out, simdjson::simdjson_result<sim
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/annotations.h" */
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */

 #include <concepts>
 #include <limits>
 #if SIMDJSON_STATIC_REFLECTION
 #include <meta>
+#include <vector>
 // #include <static_reflection> // for std::define_static_string - header not available yet
 #endif

@@ -137466,10 +175424,25 @@ error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept {

 template <std::floating_point T>
 error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept {
-  double x;
-  SIMDJSON_TRY(val.get_double().get(x));
-  out = static_cast<T>(x);
-  return SUCCESS;
+  if constexpr (std::is_same_v<T, float>) {
+    // Going through binary64 and then rounding to binary32 would round twice
+    // and could produce a value that is not the float nearest to the JSON
+    // number, so we parse to binary32 directly.
+    return val.get_float().get(out);
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  } else if constexpr (std::is_same_v<T, std::float32_t>) {
+    // Same reason as float.
+    float x;
+    SIMDJSON_TRY(val.get_float().get(x));
+    out = static_cast<T>(x);
+    return SUCCESS;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+  } else {
+    double x;
+    SIMDJSON_TRY(val.get_double().get(x));
+    out = static_cast<T>(x);
+    return SUCCESS;
+  }
 }

 template <std::signed_integral T>
@@ -137505,11 +175478,59 @@ template <concepts::constructible_from_string_view T, typename ValT>
 error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(std::is_nothrow_constructible_v<T, std::string_view>) {
   std::string_view str;
   SIMDJSON_TRY(val.get_string().get(str));
-  out = T{str};
+  if constexpr (requires { out.assign(str.data(), str.size()); }) {
+    // Copy straight into out (e.g., std::string): building a temporary and
+    // move-assigning it is markedly slower.
+    out.assign(str.data(), str.size());
+  } else {
+    out = T{str};
+  }
+  return SUCCESS;
+}
+
+#if SIMDJSON_SUPPORTS_CHAR8_T
+// any C++20 char8_t string-like type (can be constructed from std::u8string_view),
+// such as std::u8string
+template <concepts::constructible_from_u8string_view T, typename ValT>
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(std::is_nothrow_constructible_v<T, std::u8string_view>) {
+  std::u8string_view str;
+  SIMDJSON_TRY(val.get_u8string().get(str));
+  if constexpr (requires { out.assign(str.data(), str.size()); }) {
+    // Copy straight into out (e.g., std::u8string), as for std::string above.
+    out.assign(str.data(), str.size());
+  } else {
+    out = T{str};
+  }
   return SUCCESS;
 }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T


+namespace details {
+// Whether to deserialize the elements of the container T directly into a new
+// element (emplace_one(out) and then get) rather than into a temporary that is
+// then moved into the container. A temporary of a trivially copyable type lives
+// in registers and is cheap to move, and value-initializing a small element in
+// place costs more than it saves (GCC zeroes it with rep stos). But moving a
+// large element, or a string (whose deserialization then needs a temporary
+// string and a move assignment of its own), is expensive.
+template <typename T>
+concept deserialize_in_place =
+    concepts::returns_reference<T> && requires(T &c) { c.pop_back(); } &&
+    !std::is_trivially_copyable_v<typename T::value_type> &&
+    (sizeof(typename T::value_type) > 32 || concepts::constructible_from_string_view<typename T::value_type>);
+
+// Removes the last element of the container on destruction while armed.
+template <typename T>
+struct pop_back_guard {
+  T &container;
+  bool armed{true};
+  ~pop_back_guard() {
+    if (armed) { container.pop_back(); }
+  }
+};
+} // namespace details
+
 /**
  * STL containers have several constructors including one that takes a single
  * size argument. Thus, some compilers (Visual Studio) will not be able to
@@ -137533,22 +175554,65 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
     SIMDJSON_TRY(val.get_array().get(arr));
   }

-  for (auto v : arr) {
-    if constexpr (concepts::returns_reference<T>) {
-      if (auto const err = v.get<value_type>().get(concepts::emplace_one(out));
-          err) {
-        // If an error occurs, the empty element that we just inserted gets
-        // removed. We're not using a temp variable because if T is a heavy
-        // type, we want the valid path to be the fast path and the slow path be
-        // the path that has errors in it.
-        if constexpr (requires { out.pop_back(); }) {
-          static_cast<void>(out.pop_back());
+  if constexpr (std::is_same_v<T, std::vector<value_type>> && !std::is_same_v<value_type, bool>) {
+    // Collect the elements in a per-thread scratch vector that keeps its
+    // capacity from call to call, then move them into out after reserving the
+    // exact size: out is allocated once instead of being regrown. A nested
+    // array of the same type finds the scratch busy and takes the paths below.
+    // Prior related work: jsonifier keeps a thread-local vector and sizes the
+    // caller's vector from that element count (parse_impl.hpp,
+    // https://github.com/nihilai-collective/Jsonifier).
+    struct scratch_space {
+      std::vector<value_type> elements{};
+      bool busy{false};
+    };
+    static thread_local scratch_space scratch;
+    if (!scratch.busy && out.empty()) {
+      struct release_scratch {
+        scratch_space &s;
+        T &out;
+        size_t parsed{0};
+        bool complete{false};
+        // On an error or an exception, out gets the elements parsed so far (as
+        // with the loops below), without allocating. Kept out of the hot path.
+        simdjson_never_inline void keep_parsed() noexcept {
+          s.elements.resize(parsed);
+          out.swap(s.elements);
         }
-        return err;
-      }
-    } else {
+        ~release_scratch() {
+          if (simdjson_unlikely(!complete)) { keep_parsed(); }
+          s.elements.clear();
+          // Do not hold on to the memory of a very large array.
+          if (s.elements.capacity() * sizeof(value_type) > (1 << 20)) { std::vector<value_type>().swap(s.elements); }
+          s.busy = false;
+        }
+      } release{scratch, out};
+      scratch.busy = true;
+      for (auto v : arr) {
+        SIMDJSON_TRY(v.get<value_type>(scratch.elements.emplace_back()));
+        release.parsed++;
+      }
+      out.reserve(release.parsed);
+      release.complete = true;
+      for (auto &e : scratch.elements) { out.emplace_back(std::move(e)); }
+      return SUCCESS;
+    }
+  }
+  if constexpr (details::deserialize_in_place<T>) {
+    for (auto v : arr) {
+      auto &slot = concepts::emplace_one(out);
+      // An error or an exception (a user tag_invoke may throw) must not leave
+      // a partially deserialized element behind.
+      details::pop_back_guard<T> guard{out};
+      SIMDJSON_TRY(v.get<value_type>(slot));
+      guard.armed = false;
+    }
+  } else {
+    for (auto v : arr) {
+      // Deserialize into a temporary first: an error or an exception (a user
+      // tag_invoke may throw) must not leave a default-constructed element behind.
       value_type temp;
-      if (auto const err = v.get<value_type>().get(temp); err) {
+      if (auto const err = v.get<value_type>(temp); err) {
         return err;
       }
       concepts::emplace_one(out, std::move(temp));
@@ -137589,7 +175653,7 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
 }

 template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, westmere::ondemand::object &obj, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, westmere::ondemand::object &obj, T &out) noexcept(false) {
   using value_type = typename std::remove_cvref_t<T>::mapped_type;

   out.clear();
@@ -137608,21 +175672,21 @@ error_code tag_invoke(deserialize_tag, westmere::ondemand::object &obj, T &out)
 }

 template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, westmere::ondemand::value &val, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, westmere::ondemand::value &val, T &out) noexcept(false) {
   westmere::ondemand::object obj;
   SIMDJSON_TRY(val.get_object().get(obj));
   return simdjson::deserialize(obj, out);
 }

 template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, westmere::ondemand::document &doc, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, westmere::ondemand::document &doc, T &out) noexcept(false) {
   westmere::ondemand::object obj;
   SIMDJSON_TRY(doc.get_object().get(obj));
   return simdjson::deserialize(obj, out);
 }

 template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, westmere::ondemand::document_reference &doc, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, westmere::ondemand::document_reference &doc, T &out) noexcept(false) {
   westmere::ondemand::object obj;
   SIMDJSON_TRY(doc.get_object().get(obj));
   return simdjson::deserialize(obj, out);
@@ -137633,10 +175697,6 @@ error_code tag_invoke(deserialize_tag, westmere::ondemand::document_reference &d
  * This CPO (Customization Point Object) will help deserialize into
  * smart pointers.
  *
- * If constructing T is nothrow, this conversion should be nothrow as well since
- * we return MEMALLOC if we're not able to allocate memory instead of throwing
- * the error message.
- *
  * @tparam T The type inside the smart pointer
  * @tparam ValT document/value type
  * @param val document/value
@@ -137644,7 +175704,7 @@ error_code tag_invoke(deserialize_tag, westmere::ondemand::document_reference &d
  * @return status of the conversion
  */
 template <concepts::smart_pointer T, typename ValT>
-error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(nothrow_deserializable<typename std::remove_cvref_t<T>::element_type, ValT>) {
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
   using element_type = typename std::remove_cvref_t<T>::element_type;

   // For better error messages, don't use these as constraints on
@@ -137656,12 +175716,13 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(nothrow_deser
       std::is_default_constructible_v<element_type>,
       "The specified type inside the unique_ptr must default constructible.");

-  auto ptr = new (std::nothrow) element_type();
-  if (ptr == nullptr) {
+  // Own the allocation before get(): a user tag_invoke may throw.
+  std::unique_ptr<element_type> ptr(new (std::nothrow) element_type());
+  if (!ptr) {
     return MEMALLOC;
   }
   SIMDJSON_TRY(val.template get<element_type>(*ptr));
-  out.reset(ptr);
+  out = std::move(ptr);
   return SUCCESS;
 }

@@ -137693,53 +175754,595 @@ error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept(nothrow_deser

 template <typename T>
 constexpr bool user_defined_type = (std::is_class_v<T>
-&& !std::is_same_v<T, std::string> && !std::is_same_v<T, std::string_view> && !concepts::optional_type<T> &&
-!concepts::appendable_containers<T>);
+&& !std::is_same_v<T, std::string> && !std::is_same_v<T, std::string_view>
+#if SIMDJSON_SUPPORTS_CHAR8_T
+// The char8_t string types are class types with no reflectable members, so
+// without this they would be taken for user structs and the reflection
+// overload below would compete with the u8 string overload in
+// std_deserialize.h, making every tag_invoke call on them ambiguous.
+&& !std::is_same_v<T, std::u8string> && !std::is_same_v<T, std::u8string_view>
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+&& !concepts::optional_type<T> &&
+!concepts::appendable_containers<T>
+// simdjson's own types (array, object, value, raw_json_string, number,
+// document, document_reference) have dedicated get<T>() specializations and
+// must never go through reflection.
+&& !is_builtin_deserializable_v<T>
+&& !std::is_same_v<T, westmere::ondemand::number>
+&& !std::is_same_v<T, westmere::ondemand::document>
+&& !std::is_same_v<T, westmere::ondemand::document_reference>);
+
+
+// key_selector_reflection_detail is defined unconditionally (it only requires
+// static reflection). It provides the compile-time machinery for building a
+// key_selector from a struct's members, the per-member helpers shared by every
+// deserialization path (they implement the annotations of annotations.h), the
+// ordered per-member path used by the opt-out build (see
+// deserialize_struct_ordered below), and the scan used for structs annotated
+// with deny_unknown_fields and as an automatic fallback (see
+// deserialize_struct_scan and keys_fit_selector below).
+namespace key_selector_reflection_detail {
+
+// A member participates if it is public, non-const, and not annotated with skip
+// or skip_deserializing.
+consteval bool is_eligible_member(std::meta::info mem) {
+  return !std::meta::is_const(mem) && std::meta::is_public(mem)
+      && !simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_tag)
+      && !simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_deserializing_tag);
+}
+
+// A field is a data member reached from T through a chain of members: usually a
+// direct member of T (a path of length one), or a member of a member annotated
+// with flatten (the members of a flattened structure are fields of the enclosing
+// one). A field is identified by the type member_path<m1, m2, ..., leaf>.
+template <std::meta::info First, std::meta::info... Rest>
+struct member_path {
+  // The data member holding the value; its annotations drive (de)serialization.
+  static constexpr std::meta::info leaf = [] {
+    std::meta::info members[] = {First, Rest...};
+    return members[sizeof...(Rest)];
+  }();
+  template <typename T>
+  static simdjson_inline constexpr auto &get(T &obj) noexcept {
+    if constexpr (sizeof...(Rest) == 0) {
+      return obj.[:First:];
+    } else {
+      return member_path<Rest...>::get(obj.[:First:]);
+    }
+  }
+};
+
+// True when T can be flattened: a structure deserialized member by member (not
+// a string, a container, an optional, a smart pointer, ...).
+template <typename T>
+constexpr bool flattenable_type = user_defined_type<T> && !concepts::string_view_keyed_map<T>
+    && !concepts::container_but_not_string<T> && !concepts::smart_pointer<T>;
+
+consteval void append_eligible_fields(std::meta::info type, std::vector<std::meta::info> &prefix,
+                                      std::vector<std::meta::info> &fields) {
+  for (std::meta::info mem : std::meta::nonstatic_data_members_of(type, std::meta::access_context::unchecked())) {
+    if (!is_eligible_member(mem)) { continue; }
+    prefix.push_back(std::meta::reflect_constant(mem));
+    if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::flatten_tag)) {
+      std::meta::info flattened = simdjson::detail::flattened_type(mem);
+      if (!std::meta::extract<bool>(std::meta::substitute(^^flattenable_type, {flattened}))) {
+        throw std::meta::exception(u8"simdjson::flatten requires a member whose type is a structure deserialized member by member", mem);
+      }
+      append_eligible_fields(flattened, prefix, fields);
+    } else {
+      fields.push_back(std::meta::substitute(^^member_path, prefix));
+    }
+    prefix.pop_back();
+  }
+}
+
+// The fields of `type` that participate in deserialization, in declaration order
+// (the fields of a flattened member take its place). A field's position in this
+// list is its "field index" below.
+consteval std::vector<std::meta::info> eligible_fields(std::meta::info type) {
+  std::vector<std::meta::info> prefix;
+  std::vector<std::meta::info> fields;
+  append_eligible_fields(type, prefix, fields);
+  return fields;
+}
+
+// The data member at the end of a field path.
+consteval std::meta::info field_leaf(std::meta::info path) {
+  return std::meta::extract<std::meta::info>(std::meta::template_arguments_of(path).back());
+}
+
+// Number of fields that participate in deserialization. A class can have zero
+// eligible fields (e.g. std::chrono::time_point, whose only data member is
+// private): an empty key_selector cannot be built, so the tag_invoke below
+// special-cases this count.
+template <typename T>
+consteval std::size_t eligible_field_count() {
+  return eligible_fields(^^T).size();
+}
+
+// The keys accepted for a member (or an enumerator): its JSON key followed by its
+// aliases. They are static strings, so they can be used as template arguments.
+consteval std::vector<const char *> accepted_keys_of(std::meta::info entity) {
+  std::vector<const char *> keys;
+  for (std::string_view key : simdjson::detail::json_key_names(entity)) {
+    bool repeated = false;
+    for (const char *previous : keys) { repeated = repeated || std::string_view(previous) == key; }
+    if (!repeated) { keys.push_back(std::define_static_string(key)); }
+  }
+  return keys;
+}
+
+// Every key accepted for T: the eligible fields in order, each followed by its
+// aliases. This is the key list of T's key_selector.
+consteval std::vector<const char *> accepted_keys(std::meta::info type) {
+  std::vector<const char *> keys;
+  for (std::meta::info path : eligible_fields(type)) {
+    for (const char *key : accepted_keys_of(field_leaf(path))) { keys.push_back(key); }
+  }
+  return keys;
+}
+
+// The field index of each entry of accepted_keys(type).
+consteval std::vector<std::size_t> accepted_key_fields(std::meta::info type) {
+  std::vector<std::size_t> key_fields;
+  std::vector<std::meta::info> fields = eligible_fields(type);
+  for (std::size_t i = 0; i < fields.size(); ++i) {
+    for (std::size_t j = 0; j < accepted_keys_of(field_leaf(fields[i])).size(); ++j) { key_fields.push_back(i); }
+  }
+  return key_fields;
+}
+
+// True when no two eligible fields of T accept the same key. Otherwise one JSON
+// key would have to fill several members: this is reported at compile time.
+template <typename T>
+consteval bool accepted_keys_are_distinct() {
+  std::vector<const char *> keys = accepted_keys(^^T);
+  for (std::size_t i = 0; i < keys.size(); ++i) {
+    for (std::size_t j = i + 1; j < keys.size(); ++j) {
+      if (std::string_view(keys[i]) == std::string_view(keys[j])) { return false; }
+    }
+  }
+  return true;
+}
+
+// True when some accepted key of T is written with escape sequences in JSON (a
+// double quote, a backslash or a control character). Such a key can only be
+// matched by comparing unescaped keys (deserialize_struct_scan): obj[key] and
+// the key_selector compare the raw bytes.
+template <typename T>
+consteval bool keys_need_unescaping() {
+  for (std::string_view key : accepted_keys(^^T)) {
+    for (char c : key) {
+      if (c == '\\' || c == '"' || static_cast<unsigned char>(c) < 0x20) { return true; }
+    }
+  }
+  return false;
+}

+// True when some eligible field of T has aliases: several selector keys may then
+// map to the same field.
+template <typename T>
+consteval bool has_aliases() {
+  return accepted_keys(^^T).size() != eligible_fields(^^T).size();
+}
+
+// True when `mem` is annotated with default_value or default_from (directly, or
+// through a default_value annotation on the enclosing structure).
+template <auto mem>
+consteval bool has_default() {
+  return simdjson::detail::has_annotation(mem, ^^simdjson::detail::default_value_tag)
+      || simdjson::detail::has_annotation(std::meta::parent_of(mem), ^^simdjson::detail::default_value_tag)
+      || simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::default_from_t) != std::meta::info{};
+}
+
+// True when a missing key for `mem` is not an error: optional members and
+// members with a default.
+template <auto mem>
+consteval bool may_be_absent() {
+  return concepts::optional_type<typename [: std::meta::type_of(mem) :]> || has_default<mem>();
+}
+
+// True when every eligible field of T is required (none may be absent). In that
+// case presence can be checked with a single match count instead of a per-field
+// "seen" array.
+template <typename T>
+consteval bool all_eligible_fields_required() {
+  bool all_required = true;
+  template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+    if constexpr (may_be_absent<[: path :]::leaf>()) {
+      all_required = false;
+    }
+  }
+  return all_required;
+}
+
+// `key` as a constevalutil::fixed_string usable as an NTTP.
+template <const char *key>
+consteval auto key_fixed_string() {
+  constexpr std::string_view key_view{ key };
+  char buffer[key_view.size() + 1] = {};
+  for (std::size_t i = 0; i < key_view.size(); ++i) { buffer[i] = key_view[i]; }
+  return constevalutil::fixed_string<key_view.size() + 1>(buffer);
+}
+
+// key_selector template arguments (one fixed_string per accepted key), in the
+// order of accepted_keys.
+template <typename T>
+consteval std::vector<std::meta::info> selector_key_args() {
+  std::vector<std::meta::info> args;
+  template for (constexpr const char *key : std::define_static_array(accepted_keys(^^T))) {
+    args.push_back(std::meta::reflect_constant(key_fixed_string<key>()));
+  }
+  return args;
+}
+
+// key_selector whose keys are exactly T's accepted keys (index i <-> i-th key).
+template <typename T>
+using selector_for = typename [: std::meta::substitute(
+    ^^westmere::ondemand::key_selector, selector_key_args<T>()) :];
+
+// True when T's accepted keys satisfy every key_selector requirement, so a
+// key_selector can be built for T without a compile-time error. This mirrors the
+// key_selector limits (see key_selector.h): at most 255 keys, each key non-empty
+// and at most 63 characters, no backslash / double-quote / null byte, and all
+// keys distinct. Keys with any other control character are excluded too: they
+// are escaped in JSON, so their raw bytes never match. When this returns false
+// the deserializer falls back to deserialize_struct_scan instead of failing to
+// compile.
+template <typename T>
+consteval bool keys_fit_selector() {
+  std::vector<std::string_view> keys;
+  for (const char *key : accepted_keys(^^T)) { keys.push_back(key); }
+  if (keys.size() > 255) { return false; }
+  for (std::size_t i = 0; i < keys.size(); ++i) {
+    if (keys[i].empty() || keys[i].size() > 63) { return false; }
+    for (char c : keys[i]) {
+      if (c == '\\' || c == '"' || static_cast<unsigned char>(c) < 0x20) { return false; }
+    }
+    for (std::size_t j = i + 1; j < keys.size(); ++j) {
+      if (keys[i] == keys[j]) { return false; }
+    }
+  }
+  return true;
+}
+
+// True when the class `adapter` declares a member named deserialize.
+consteval bool declares_deserialize(std::meta::info adapter) {
+  for (std::meta::info m : std::meta::members_of(adapter, std::meta::access_context::unchecked())) {
+    if (std::meta::has_identifier(m) && std::meta::identifier_of(m) == "deserialize") { return true; }
+  }
+  return false;
+}
+
+// Deserialize a JSON value into `target`, the storage of member `mem`, through
+// the member's with<Adapter> annotation when the adapter provides a deserialize
+// function.
+template <auto mem, typename ValueT, typename M>
+simdjson_warn_unused simdjson_inline error_code deserialize_member_value(ValueT &field_value, M &target) noexcept(false) {
+  constexpr std::meta::info with_type = simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::with_t);
+  if constexpr (with_type != std::meta::info{}) {
+    using adapter = typename [: with_type :]::adapter;
+    using ondemand_value = westmere::ondemand::value;
+    if constexpr (requires { { adapter::deserialize(field_value, target) } -> std::convertible_to<error_code>; }) {
+      return adapter::deserialize(field_value, target);
+    } else if constexpr (requires(ondemand_value &v) { { adapter::deserialize(v, target) } -> std::convertible_to<error_code>; }
+                         && requires { field_value.get_value(); }) {
+      // A transparent structure read from a document: the adapter takes an
+      // ondemand::value. A scalar document cannot be viewed as a value, so it
+      // reports SCALAR_DOCUMENT_AS_VALUE (an adapter taking auto& receives the
+      // document itself and has no such limitation).
+      ondemand_value v;
+      SIMDJSON_TRY(field_value.get_value().get(v));
+      return adapter::deserialize(v, target);
+    } else {
+      static_assert(!declares_deserialize(^^adapter),
+                    "the deserialize function of a simdjson::with adapter must be callable as "
+                    "Adapter::deserialize(simdjson::ondemand::value &, T &) and return an error_code");
+      return field_value.get(target);
+    }
+  } else {
+    return field_value.get(target);
+  }
+}
+
+// Deserialize the storage `target` of member `mem` from a JSON value.
+template <auto mem, typename ValueT, typename M>
+simdjson_warn_unused simdjson_inline error_code deserialize_member(ValueT &field_value, M &target) noexcept(false) {
+  if constexpr (has_default<mem>() && std::is_default_constructible_v<M> && std::is_move_assignable_v<M>) {
+    // A present key replaces the default value: deserialize into a fresh
+    // temporary so that, e.g., a container does not append to its default
+    // content, and a failure leaves the default untouched.
+    M value{};
+    SIMDJSON_TRY(deserialize_member_value<mem>(field_value, value));
+    target = std::move(value);
+    return SUCCESS;
+  } else {
+    return deserialize_member_value<mem>(field_value, target);
+  }
+}

+// Deserialize the field with the given field index.
+template <typename T>
+simdjson_warn_unused simdjson_inline error_code deserialize_field_at(
+    std::size_t field_index, westmere::ondemand::value field_value, T &out) noexcept(false) {
+  std::size_t counter = 0;
+  template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+    using field = [: path :];
+    if (field_index == counter) { return deserialize_member<field::leaf>(field_value, field::get(out)); }
+    ++counter;
+  }
+  return SUCCESS;
+}
+
+// Called when the key(s) of member `mem`, stored in `target`, are missing from
+// the JSON object: an error for a required member, a call to the default_from
+// factory, or nothing at all.
+template <auto mem, typename M>
+simdjson_warn_unused simdjson_inline error_code on_missing_member(M &target) noexcept(false) {
+  constexpr std::meta::info default_from_type = simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::default_from_t);
+  if constexpr (default_from_type != std::meta::info{}) {
+    target = [: default_from_type :]::factory();
+    return SUCCESS;
+  } else if constexpr (may_be_absent<mem>()) {
+    // For optional and default_value members, a missing key is not an error:
+    // leave the member at its current (default) value.
+    (void)target;
+    return SUCCESS;
+  } else {
+    (void)target;
+    return NO_SUCH_FIELD;
+  }
+}
+
+// Report or handle every field that was not seen in the JSON object.
+template <typename T, std::size_t N>
+simdjson_warn_unused simdjson_inline error_code handle_missing_fields(
+    const std::array<bool, N> &seen_field, T &out) noexcept(false) {
+  std::size_t counter = 0;
+  template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+    using field = [: path :];
+    if (!seen_field[counter]) { SIMDJSON_TRY(on_missing_member<field::leaf>(field::get(out))); }
+    ++counter;
+  }
+  return SUCCESS;
+}
+
+// Ordered, per-field deserialization: one obj[key] lookup per eligible field
+// (and per alias, until one is found). This is the opt-out path
+// (-DSIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION=1), except for structs with keys
+// that need unescaping (see keys_need_unescaping).
+template <typename T>
+simdjson_warn_unused error_code deserialize_struct_ordered(
+    westmere::ondemand::object &obj, T &out) noexcept(false) {
+  template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+    using field = [: path :];
+    westmere::ondemand::value field_value;
+    error_code error = NO_SUCH_FIELD;
+    template for (constexpr const char *key : std::define_static_array(accepted_keys_of(field::leaf))) {
+      if (error == NO_SUCH_FIELD) { error = obj[std::string_view(key)].get(field_value); }
+    }
+    if (error == NO_SUCH_FIELD) {
+      SIMDJSON_TRY(on_missing_member<field::leaf>(field::get(out)));
+    } else if (error) {
+      return error;
+    } else {
+      SIMDJSON_TRY(deserialize_member<field::leaf>(field_value, field::get(out)));
+    }
+  }
+  return SUCCESS;
+}
+
+// Appends the JSON keys that serialization writes for members of `type` that
+// deserialization cannot assign (const or non-public members, directly or
+// through flatten). `all` is true inside a flattened member that is itself
+// unassignable. Keys of skip_deserializing members are not included: they are
+// unknown keys, as documented.
+consteval void append_unassignable_keys(std::meta::info type, bool all, std::vector<const char *> &keys) {
+  for (std::meta::info mem : std::meta::nonstatic_data_members_of(type, std::meta::access_context::unchecked())) {
+    if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_tag)
+        || simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_serializing_tag)
+        || simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_deserializing_tag)) {
+      continue;
+    }
+    bool unassignable = all || !is_eligible_member(mem);
+    if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::flatten_tag)) {
+      append_unassignable_keys(simdjson::detail::flattened_type(mem), unassignable, keys);
+    } else if (unassignable) {
+      keys.push_back(std::define_static_string(simdjson::detail::json_key_name(mem)));
+    }
+  }
+}
+
+consteval std::vector<const char *> unassignable_keys(std::meta::info type) {
+  std::vector<const char *> keys;
+  append_unassignable_keys(type, false, keys);
+  return keys;
+}
+
+// Deserialization by a single pass over every field of the object, comparing
+// unescaped keys. It is used for structs annotated with deny_unknown_fields
+// (DenyUnknown = true), where a key that does not map to an eligible field is
+// reported as UNKNOWN_FIELD, and as the fallback for structs whose keys the
+// key_selector or obj[key] cannot match (see keys_fit_selector and
+// keys_need_unescaping). As with object::for_each, the first occurrence of a
+// field wins (later duplicates, or aliases of a field already seen, are ignored).
+template <bool DenyUnknown, typename T>
+simdjson_warn_unused error_code deserialize_struct_scan(
+    westmere::ondemand::object &obj, T &out) noexcept(false) {
+  static constexpr auto keys = std::define_static_array(accepted_keys(^^T));
+  static constexpr auto key_fields = std::define_static_array(accepted_key_fields(^^T));
+  std::array<bool, eligible_field_count<T>()> seen_field{};
+  for (auto field_result : obj) {
+    westmere::ondemand::field json_field;
+    SIMDJSON_TRY(std::move(field_result).get(json_field));
+    std::string_view key;
+    SIMDJSON_TRY(json_field.unescaped_key().get(key));
+    std::size_t key_index = keys.size();
+    for (std::size_t i = 0; i < keys.size(); ++i) {
+      if (key == std::string_view(keys[i])) { key_index = i; break; }
+    }
+    if (key_index == keys.size()) {
+      if constexpr (DenyUnknown) {
+        // A key that T itself serializes (e.g. of a const member) is not
+        // unknown: a serialized value must parse back.
+        static constexpr auto ignored_keys = std::define_static_array(unassignable_keys(^^T));
+        bool ignored = false;
+        for (const char *ignored_key : ignored_keys) {
+          if (key == std::string_view(ignored_key)) { ignored = true; break; }
+        }
+        if (!ignored) { return UNKNOWN_FIELD; }
+      }
+      continue;
+    }
+    const std::size_t field_index = key_fields[key_index];
+    if (seen_field[field_index]) { continue; }
+    seen_field[field_index] = true;
+    SIMDJSON_TRY(deserialize_field_at(field_index, json_field.value(), out));
+  }
+  return handle_missing_fields(seen_field, out);
+}
+
+template <typename T>
+consteval bool is_transparent() {
+  return simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag);
+}
+
+} // namespace key_selector_reflection_detail
+
+// Deserialize a reflected struct. By default this builds a compile-time
+// key_selector from the struct's members and walks each object once with
+// object::for_each (perfect-hash key matching), instead of one obj[key] lookup
+// per member. Other paths are used instead:
+//   - globally, the ordered per-member path when defining
+//     -DSIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION=1;
+//   - automatically and per-type, a scan of the object comparing unescaped keys
+//     when the struct's keys do not fit the key_selector limits (see
+//     keys_fit_selector) or cannot be compared raw (see keys_need_unescaping),
+//     so that long member names and the like keep compiling rather than
+//     tripping a static_assert.
+// Structs annotated with deny_unknown_fields always use the strict scan, and
+// structs annotated with transparent are deserialized as their single member.
+//
+// noexcept(false): a member's tag_invoke may throw and the exception must reach
+// the caller of get<T>().
 template <typename T, typename ValT>
   requires(user_defined_type<T> && std::is_class_v<T>)
-error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
+  if constexpr (key_selector_reflection_detail::is_transparent<T>()) {
+    constexpr auto mem = simdjson::detail::transparent_member(^^T);
+    if constexpr (std::is_same_v<std::remove_cvref_t<ValT>, westmere::ondemand::object>) {
+      // We were handed an object: only a structure can be deserialized from it.
+      if constexpr (user_defined_type<typename [: std::meta::type_of(mem) :]>) {
+        return tag_invoke(deserialize_tag{}, val, out.[:mem:]);
+      } else {
+        return INCORRECT_TYPE;
+      }
+    } else {
+      return key_selector_reflection_detail::deserialize_member<mem>(val, out.[:mem:]);
+    }
+  } else {
+  static_assert(key_selector_reflection_detail::accepted_keys_are_distinct<T>(),
+                "two members of this structure accept the same JSON key (check rename, alias, "
+                "rename_all and flatten)");
   westmere::ondemand::object obj;
   if constexpr (std::is_same_v<std::remove_cvref_t<ValT>, westmere::ondemand::object>) {
     obj = val;
   } else {
     SIMDJSON_TRY(val.get_object().get(obj));
   }
-  template for (constexpr auto mem : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
-    if constexpr (!std::meta::is_const(mem) && std::meta::is_public(mem)) {
-      constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
-      if constexpr (concepts::optional_type<decltype(out.[:mem:])>) {
-        // for optional members, it's ok if the key is missing
-        auto error = obj[key].get(out.[:mem:]);
-        if (error && error != NO_SUCH_FIELD) {
-          if(error == NO_SUCH_FIELD) {
-            out.[:mem:].reset();
-            continue;
-          }
-          return error;
-        }
-      } else {
-        // for non-optional members, the key must be present
-        SIMDJSON_TRY(obj[key].get(out.[:mem:]));
+  if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::deny_unknown_fields_tag)) {
+    return key_selector_reflection_detail::deserialize_struct_scan<true>(obj, out);
+  } else {
+#if defined(SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION) && SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION
+  // Opt-out build: use the ordered per-member path, unless obj[key] cannot
+  // match T's keys.
+  if constexpr (key_selector_reflection_detail::keys_need_unescaping<T>()) {
+    return key_selector_reflection_detail::deserialize_struct_scan<false>(obj, out);
+  } else {
+    return key_selector_reflection_detail::deserialize_struct_ordered(obj, out);
+  }
+#else
+  if constexpr (key_selector_reflection_detail::eligible_field_count<T>() == 0) {
+    // No fields to deserialize: an empty key_selector cannot be built, so just
+    // validate that the input is an object (done above) and succeed. Mirrors the
+    // ordered per-member path, which iterates over zero members.
+    (void)out;
+    (void)obj;
+    return SUCCESS;
+  } else if constexpr (!key_selector_reflection_detail::keys_fit_selector<T>()) {
+    // Automatic fallback: T's accepted keys do not fit the key_selector limits
+    // (e.g. a member name longer than 63 characters, or a key with a double
+    // quote), so building a selector would be a compile error. Scan the object
+    // instead, so the default never breaks a struct that the opt-out path would
+    // accept.
+    return key_selector_reflection_detail::deserialize_struct_scan<false>(obj, out);
+  } else {
+  using selector = key_selector_reflection_detail::selector_for<T>;
+  if constexpr (key_selector_reflection_detail::all_eligible_fields_required<T>()
+                && !key_selector_reflection_detail::has_aliases<T>()) {
+    // Fast path: every member is required and has a single key. A single
+    // for_each pass parses each matched field; the returned match count then
+    // tells us whether every member was present (matched_count ==
+    // selector::size()) without a per-member "seen" array. A value-parse error
+    // (e.g. a type mismatch) is propagated by for_each.
+    auto walk = obj.template for_each<selector>(
+        [&](std::size_t matched_index, westmere::ondemand::value field_value) -> error_code {
+      std::size_t counter = 0;
+      template for (constexpr auto path : std::define_static_array(key_selector_reflection_detail::eligible_fields(^^T))) {
+        using field = [: path :];
+        if (matched_index == counter) { return key_selector_reflection_detail::deserialize_member<field::leaf>(field_value, field::get(out)); }
+        ++counter;
       }
-    }
-  };
-  return simdjson::SUCCESS;
-}
-
-// Support for enum deserialization - deserialize from string representation using expand approach from P2996R12
+      return SUCCESS;
+    });
+    if (walk.error) { return walk.error; }
+    // A missing required member shows up as a short match count and is reported as
+    // NO_SUCH_FIELD, mirroring the ordered obj[key] path.
+    if (walk.matched_count != selector::size()) { return NO_SUCH_FIELD; }
+    return SUCCESS;
+  } else {
+    static constexpr auto key_fields = std::define_static_array(key_selector_reflection_detail::accepted_key_fields(^^T));
+    std::array<bool, key_selector_reflection_detail::eligible_field_count<T>()> seen_field{};
+    // Single pass over the object: each field whose key matches a member (or one
+    // of its aliases) yields its selector index, which we map back to the
+    // corresponding member. The first key seen for a member wins. The callback
+    // returns an error_code so that a value-parse error (e.g. a type mismatch on
+    // a matched field) is propagated by for_each instead of being silently dropped.
+    error_code walk_error = obj.template for_each<selector>(
+        [&](std::size_t matched_index, westmere::ondemand::value field_value) -> error_code {
+      const std::size_t field_index = key_fields[matched_index];
+      if (seen_field[field_index]) { return SUCCESS; }
+      seen_field[field_index] = true;
+      return key_selector_reflection_detail::deserialize_field_at(field_index, field_value, out);
+    });
+    if (walk_error) { return walk_error; }
+    // Required members must be present: a missing one is reported as
+    // NO_SUCH_FIELD, mirroring the ordered obj[key] path. Optional and defaulted
+    // members may be absent.
+    return key_selector_reflection_detail::handle_missing_fields(seen_field, out);
+  }
+  }
+#endif // SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION
+  }
+  }
+}
+
+// Support for enum deserialization - deserialize from string representation using
+// expand approach from P2996R12. The accepted strings are the enumerator's JSON key
+// (see rename and rename_all) and its aliases.
 template <typename T, typename ValT>
   requires(std::is_enum_v<T>)
 error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
 #if SIMDJSON_STATIC_REFLECTION
   std::string_view str;
   SIMDJSON_TRY(val.get_string().get(str));
-  constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
+  static constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
   template for (constexpr auto enum_val : enumerators) {
-    if (str == std::meta::identifier_of(enum_val)) {
-      out = [:enum_val:];
-      return SUCCESS;
+    template for (constexpr const char *key : std::define_static_array(key_selector_reflection_detail::accepted_keys_of(enum_val))) {
+      if (str == std::string_view(key)) {
+        out = [:enum_val:];
+        return SUCCESS;
+      }
     }
   };

@@ -137755,33 +176358,25 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {

 template <typename simdjson_value, typename T>
   requires(user_defined_type<std::remove_cvref_t<T>>)
-error_code tag_invoke(deserialize_tag, simdjson_value &val, std::unique_ptr<T> &out) noexcept {
-  if (!out) {
-    out = std::make_unique<T>();
-    if (!out) {
-      return MEMALLOC;
-    }
-  }
-  if (auto err = val.get(*out)) {
-    out.reset();
-    return err;
+error_code tag_invoke(deserialize_tag, simdjson_value &val, std::unique_ptr<T> &out) noexcept(false) {
+  std::unique_ptr<T> ptr(new (std::nothrow) T());
+  if (!ptr) {
+    return MEMALLOC;
   }
+  SIMDJSON_TRY(val.get(*ptr));
+  out = std::move(ptr);
   return SUCCESS;
 }

 template <typename simdjson_value, typename T>
   requires(user_defined_type<std::remove_cvref_t<T>>)
-error_code tag_invoke(deserialize_tag, simdjson_value &val, std::shared_ptr<T> &out) noexcept {
-  if (!out) {
-    out = std::make_shared<T>();
-    if (!out) {
-      return MEMALLOC;
-    }
-  }
-  if (auto err = val.get(*out)) {
-    out.reset();
-    return err;
+error_code tag_invoke(deserialize_tag, simdjson_value &val, std::shared_ptr<T> &out) noexcept(false) {
+  std::shared_ptr<T> ptr(new (std::nothrow) T());
+  if (!ptr) {
+    return MEMALLOC;
   }
+  SIMDJSON_TRY(val.get(*ptr));
+  out = std::move(ptr);
   return SUCCESS;
 }

@@ -138093,9 +176688,17 @@ simdjson_inline simdjson_result<array> array::started(value_iterator &iter) noex
   return array(iter);
 }

-simdjson_inline simdjson_result<array_iterator> array::begin() noexcept {
+simdjson_inline simdjson_result<array_iterator> array::begin() & noexcept {
 #if SIMDJSON_DEVELOPMENT_CHECKS
-  if (!iter.is_at_iterator_start()) { return OUT_OF_ORDER_ITERATION; }
+  if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
+  return array_iterator(iter, this);
+#endif
+  return array_iterator(iter);
+}
+simdjson_inline simdjson_result<array_iterator> array::begin() && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  // The array is a temporary that the iterator may outlive: do not lock it.
+  if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
 #endif
   return array_iterator(iter);
 }
@@ -138122,6 +176725,9 @@ simdjson_inline simdjson_result<std::string_view> array::raw_json() noexcept {
 SIMDJSON_PUSH_DISABLE_WARNINGS
 SIMDJSON_DISABLE_STRICT_OVERFLOW_WARNING
 simdjson_inline simdjson_result<size_t> array::count_elements() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   size_t count{0};
   // Important: we do not consume any of the values.
   for(simdjson_unused auto v : *this) { count++; }
@@ -138135,6 +176741,9 @@ simdjson_inline simdjson_result<size_t> array::count_elements() & noexcept {
 SIMDJSON_POP_DISABLE_WARNINGS

 simdjson_inline simdjson_result<bool> array::is_empty() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   bool is_not_empty;
   auto error = iter.reset_array().get(is_not_empty);
   if(error) { return error; }
@@ -138142,31 +176751,30 @@ simdjson_inline simdjson_result<bool> array::is_empty() & noexcept {
 }

 inline simdjson_result<bool> array::reset() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   return iter.reset_array();
 }

+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline void array::set_locked(bool _locked) noexcept {
+  locked = _locked;
+}
+#endif
+
 inline simdjson_result<value> array::at_pointer(std::string_view json_pointer) noexcept {
-  if (json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
+  // An empty pointer has no json_pointer[0]: with a default-constructed
+  // std::string_view, reading it dereferences a null pointer.
+  if (json_pointer.empty() || json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
   json_pointer = json_pointer.substr(1);
   // - means "the append position" or "the element after the end of the array"
   // We don't support this, because we're returning a real element, not a position.
   if (json_pointer == "-") { return INDEX_OUT_OF_BOUNDS; }

-  // Read the array index
   size_t array_index = 0;
   size_t i;
-  for (i = 0; i < json_pointer.length() && json_pointer[i] != '/'; i++) {
-    uint8_t digit = uint8_t(json_pointer[i] - '0');
-    // Check for non-digit in array index. If it's there, we're trying to get a field in an object
-    if (digit > 9) { return INCORRECT_TYPE; }
-    array_index = array_index*10 + digit;
-  }
-
-  // 0 followed by other digits is invalid
-  if (i > 1 && json_pointer[0] == '0') { return INVALID_JSON_POINTER; } // "JSON pointer array index has other characters after 0"
-
-  // Empty string is invalid; so is a "/" with no digits before it
-  if (i == 0) { return INVALID_JSON_POINTER; } // "Empty string in JSON pointer array index"
+  SIMDJSON_TRY(internal::parse_json_pointer_array_index(json_pointer, array_index, i));
   // Get the child
   auto child = at(array_index);
   // If there is an error, it ends here
@@ -138240,6 +176848,9 @@ inline error_code array::for_each_at_path_with_wildcard(std::string_view json_pa
 }

 simdjson_inline simdjson_result<value> array::at(size_t index) noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   size_t i = 0;
   for (auto value : *this) {
     if (i == index) { return value; }
@@ -138269,10 +176880,14 @@ simdjson_inline simdjson_result<westmere::ondemand::array>::simdjson_result(
 {
 }

-simdjson_inline simdjson_result<westmere::ondemand::array_iterator> simdjson_result<westmere::ondemand::array>::begin() noexcept {
+simdjson_inline simdjson_result<westmere::ondemand::array_iterator> simdjson_result<westmere::ondemand::array>::begin() & noexcept {
   if (error()) { return error(); }
   return first.begin();
 }
+simdjson_inline simdjson_result<westmere::ondemand::array_iterator> simdjson_result<westmere::ondemand::array>::begin() && noexcept {
+  if (error()) { return error(); }
+  return std::move(first).begin();
+}
 simdjson_inline simdjson_result<westmere::ondemand::array_iterator> simdjson_result<westmere::ondemand::array>::end() noexcept {
   if (error()) { return error(); }
   return first.end();
@@ -138335,6 +176950,59 @@ simdjson_inline array_iterator::array_iterator(const value_iterator &_iter) noex
   : iter{_iter}
 {}

+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline array_iterator::array_iterator(const value_iterator &_iter, array* _parent) noexcept
+  : parent{_parent}, iter{_iter}
+{
+  if (parent) parent->set_locked(true);
+}
+
+simdjson_inline array_iterator::~array_iterator() noexcept
+{
+  if (parent) parent->set_locked(false);
+}
+
+simdjson_inline array_iterator::array_iterator(array_iterator&& other) noexcept
+  : has_been_referenced{other.has_been_referenced},
+    parent{other.parent},
+    iter{std::move(other.iter)}
+{
+  other.parent = nullptr;
+}
+
+simdjson_inline array_iterator& array_iterator::operator=(array_iterator&& other) noexcept {
+  if (this != &other)
+  {
+    if (parent)
+      parent->set_locked(false);
+    has_been_referenced = other.has_been_referenced;
+    parent = other.parent;
+    iter = std::move(other.iter);
+
+    other.parent = nullptr;
+  }
+  return *this;
+}
+
+simdjson_inline array_iterator::array_iterator(const array_iterator& other) noexcept
+  : has_been_referenced{other.has_been_referenced},
+    parent{nullptr},
+    iter{other.iter}
+{}
+
+simdjson_inline array_iterator& array_iterator::operator=(const array_iterator& other) noexcept {
+  if (this != &other)
+  {
+    if (parent)
+      parent->set_locked(false);
+    has_been_referenced = other.has_been_referenced;
+    parent = nullptr;
+    iter = other.iter;
+  }
+  return *this;
+}
+#endif
+
 simdjson_inline simdjson_result<value> array_iterator::operator*() noexcept {
 #if SIMDJSON_DEVELOPMENT_CHECKS
    SIMDJSON_ASSUME(!has_been_referenced);
@@ -138430,6 +177098,41 @@ namespace simdjson {
 namespace westmere {
 namespace ondemand {

+/**
+ * @private Narrows the result of get_uint64() to a smaller unsigned integer type T.
+ * @returns NUMBER_OUT_OF_RANGE if the value does not fit in a T.
+ */
+template <typename T>
+simdjson_inline simdjson_result<T> narrow_integer(simdjson_result<uint64_t> wide) noexcept {
+  uint64_t result;
+  SIMDJSON_TRY(std::move(wide).get(result));
+  if (result > static_cast<uint64_t>((std::numeric_limits<T>::max)())) { return NUMBER_OUT_OF_RANGE; }
+  return static_cast<T>(result);
+}
+
+/**
+ * @private Narrows the result of get_int64() to a smaller signed integer type T.
+ * @returns NUMBER_OUT_OF_RANGE if the value does not fit in a T.
+ */
+template <typename T>
+simdjson_inline simdjson_result<T> narrow_integer(simdjson_result<int64_t> wide) noexcept {
+  int64_t result;
+  SIMDJSON_TRY(std::move(wide).get(result));
+  if (result > static_cast<int64_t>((std::numeric_limits<T>::max)()) || result < static_cast<int64_t>((std::numeric_limits<T>::min)())) { return NUMBER_OUT_OF_RANGE; }
+  return static_cast<T>(result);
+}
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+// get_float32() parses with get_float(), so float must be binary32.
+static_assert(std::numeric_limits<float>::is_iec559 && std::numeric_limits<float>::digits == 24,
+              "get_float32() requires float to be IEEE binary32");
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+// get_float64() parses with get_double(), so double must be binary64.
+static_assert(std::numeric_limits<double>::is_iec559 && std::numeric_limits<double>::digits == 53,
+              "get_float64() requires double to be IEEE binary64");
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
+
 simdjson_inline value::value(const value_iterator &_iter) noexcept
   : iter{_iter}
 {
@@ -138461,6 +177164,13 @@ simdjson_inline simdjson_result<raw_json_string> value::get_raw_json_string() no
 simdjson_inline simdjson_result<std::string_view> value::get_string(bool allow_replacement) noexcept {
   return iter.get_string(allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> value::get_u8string(bool allow_replacement) noexcept {
+  std::string_view content;
+  SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+  return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code value::get_string(string_type& receiver, bool allow_replacement) noexcept {
   return iter.get_string(receiver, allow_replacement);
@@ -138474,6 +177184,12 @@ simdjson_inline simdjson_result<double> value::get_double() noexcept {
 simdjson_inline simdjson_result<double> value::get_double_in_string() noexcept {
   return iter.get_double_in_string();
 }
+simdjson_inline simdjson_result<float> value::get_float() noexcept {
+  return iter.get_float();
+}
+simdjson_inline simdjson_result<float> value::get_float_in_string() noexcept {
+  return iter.get_float_in_string();
+}
 simdjson_inline simdjson_result<uint64_t> value::get_uint64() noexcept {
   return iter.get_uint64();
 }
@@ -138487,17 +177203,37 @@ simdjson_inline simdjson_result<int64_t> value::get_int64_in_string() noexcept {
   return iter.get_int64_in_string();
 }
 simdjson_inline simdjson_result<uint32_t> value::get_uint32() noexcept {
-  uint64_t result;
-  SIMDJSON_TRY(get_uint64().get(result));
-  if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<uint32_t>(result);
+  return narrow_integer<uint32_t>(get_uint64());
 }
 simdjson_inline simdjson_result<int32_t> value::get_int32() noexcept {
-  int64_t result;
-  SIMDJSON_TRY(get_int64().get(result));
-  if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<int32_t>(result);
+  return narrow_integer<int32_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint16_t> value::get_uint16() noexcept {
+  return narrow_integer<uint16_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int16_t> value::get_int16() noexcept {
+  return narrow_integer<int16_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint8_t> value::get_uint8() noexcept {
+  return narrow_integer<uint8_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int8_t> value::get_int8() noexcept {
+  return narrow_integer<int8_t>(get_int64());
 }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> value::get_float32() noexcept {
+  float result;
+  SIMDJSON_TRY(get_float().get(result));
+  return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> value::get_float64() noexcept {
+  double result;
+  SIMDJSON_TRY(get_double().get(result));
+  return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<bool> value::get_bool() noexcept {
   return iter.get_bool();
 }
@@ -138509,12 +177245,26 @@ template<> simdjson_inline simdjson_result<array> value::get() noexcept { return
 template<> simdjson_inline simdjson_result<object> value::get() noexcept { return get_object(); }
 template<> simdjson_inline simdjson_result<raw_json_string> value::get() noexcept { return get_raw_json_string(); }
 template<> simdjson_inline simdjson_result<std::string_view> value::get() noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> value::get() noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_inline simdjson_result<number> value::get() noexcept { return get_number(); }
 template<> simdjson_inline simdjson_result<double> value::get() noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> value::get() noexcept { return get_float(); }
 template<> simdjson_inline simdjson_result<uint64_t> value::get() noexcept { return get_uint64(); }
 template<> simdjson_inline simdjson_result<int64_t> value::get() noexcept { return get_int64(); }
 template<> simdjson_inline simdjson_result<uint32_t> value::get() noexcept { return get_uint32(); }
 template<> simdjson_inline simdjson_result<int32_t> value::get() noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> value::get() noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> value::get() noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> value::get() noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> value::get() noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> value::get() noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> value::get() noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<> simdjson_inline simdjson_result<bool> value::get() noexcept { return get_bool(); }


@@ -138522,12 +177272,26 @@ template<> simdjson_warn_unused simdjson_inline error_code value::get(array& out
 template<> simdjson_warn_unused simdjson_inline error_code value::get(object& out) noexcept { return get_object().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(raw_json_string& out) noexcept { return get_raw_json_string().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(std::string_view& out) noexcept { return get_string(false).get(out); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::u8string_view& out) noexcept { return get_u8string(false).get(out); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_warn_unused simdjson_inline error_code value::get(number& out) noexcept { return get_number().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(double& out) noexcept { return get_double().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(float& out) noexcept { return get_float().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(uint64_t& out) noexcept { return get_uint64().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(int64_t& out) noexcept { return get_int64().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(uint32_t& out) noexcept { return get_uint32().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(int32_t& out) noexcept { return get_int32().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(uint16_t& out) noexcept { return get_uint16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(int16_t& out) noexcept { return get_int16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(uint8_t& out) noexcept { return get_uint8().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(int8_t& out) noexcept { return get_int8().get(out); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::float32_t& out) noexcept { return get_float32().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::float64_t& out) noexcept { return get_float64().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<>  simdjson_warn_unused simdjson_inline error_code value::get(bool& out) noexcept { return get_bool().get(out); }

 #if SIMDJSON_EXCEPTIONS
@@ -138696,6 +177460,9 @@ inline bool is_pointer_well_formed(std::string_view json_pointer) noexcept {
 }

 simdjson_inline simdjson_result<value> value::at_pointer(std::string_view json_pointer) noexcept {
+  // The empty JSON Pointer refers to the whole value (RFC 6901), as in
+  // document::at_pointer.
+  if (json_pointer.empty()) { return value(iter); }
   json_type t;
   SIMDJSON_TRY(type().get(t));
   switch (t)
@@ -138733,6 +177500,10 @@ template <typename Func>
 template <typename Func>
 #endif
 inline error_code value::for_each_at_path_with_wildcard(std::string_view json_path, Func&& callback) noexcept {
+  // Every recursive step of for_each_at_path_with_wildcard goes through this
+  // function, and each one descends one level into the document. A path with
+  // many segments applied to a deeply nested document would otherwise recurse
+  // without bound and overflow the stack.
   if (size_t(iter.depth()) >= iter.json_iter().parser->max_depth()) { return DEPTH_ERROR; }
   json_type t;
   SIMDJSON_TRY(type().get(t));
@@ -138846,10 +177617,46 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<westmere::ondemand::val
   if (error()) { return error(); }
   return first.get_int32();
 }
+simdjson_inline simdjson_result<uint16_t> simdjson_result<westmere::ondemand::value>::get_uint16() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<westmere::ondemand::value>::get_int16() noexcept {
+  if (error()) { return error(); }
+  return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<westmere::ondemand::value>::get_uint8() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<westmere::ondemand::value>::get_int8() noexcept {
+  if (error()) { return error(); }
+  return first.get_int8();
+}
 simdjson_inline simdjson_result<double> simdjson_result<westmere::ondemand::value>::get_double() noexcept {
   if (error()) { return error(); }
   return first.get_double();
 }
+simdjson_inline simdjson_result<float> simdjson_result<westmere::ondemand::value>::get_float() noexcept {
+  if (error()) { return error(); }
+  return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<westmere::ondemand::value>::get_float_in_string() noexcept {
+  if (error()) { return error(); }
+  return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<westmere::ondemand::value>::get_float32() noexcept {
+  if (error()) { return error(); }
+  return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<westmere::ondemand::value>::get_float64() noexcept {
+  if (error()) { return error(); }
+  return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<double> simdjson_result<westmere::ondemand::value>::get_double_in_string() noexcept {
   if (error()) { return error(); }
   return first.get_double_in_string();
@@ -138858,6 +177665,12 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<westmere::onde
   if (error()) { return error(); }
   return first.get_string(allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<westmere::ondemand::value>::get_u8string(bool allow_replacement) noexcept {
+  if (error()) { return error(); }
+  return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_inline error_code simdjson_result<westmere::ondemand::value>::get_string(string_type& receiver, bool allow_replacement) noexcept {
   if (error()) { return error(); }
@@ -138886,11 +177699,23 @@ template<> simdjson_inline error_code simdjson_result<westmere::ondemand::value>
   return SUCCESS;
 }

-template<typename T> simdjson_inline simdjson_result<T> simdjson_result<westmere::ondemand::value>::get() noexcept {
+template<typename T> simdjson_inline simdjson_result<T> simdjson_result<westmere::ondemand::value>::get()
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, westmere::ondemand::value>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>();
 }
-template<typename T> simdjson_inline error_code simdjson_result<westmere::ondemand::value>::get(T &out) noexcept {
+template<typename T> simdjson_inline error_code simdjson_result<westmere::ondemand::value>::get(T &out)
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, westmere::ondemand::value>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>(out);
 }
@@ -139160,16 +177985,22 @@ simdjson_inline simdjson_result<int64_t> document::get_int64_in_string() noexcep
   return get_root_value_iterator().get_root_int64_in_string(true);
 }
 simdjson_inline simdjson_result<uint32_t> document::get_uint32() noexcept {
-  uint64_t result;
-  SIMDJSON_TRY(get_uint64().get(result));
-  if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<uint32_t>(result);
+  return narrow_integer<uint32_t>(get_uint64());
 }
 simdjson_inline simdjson_result<int32_t> document::get_int32() noexcept {
-  int64_t result;
-  SIMDJSON_TRY(get_int64().get(result));
-  if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<int32_t>(result);
+  return narrow_integer<int32_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint16_t> document::get_uint16() noexcept {
+  return narrow_integer<uint16_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int16_t> document::get_int16() noexcept {
+  return narrow_integer<int16_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint8_t> document::get_uint8() noexcept {
+  return narrow_integer<uint8_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int8_t> document::get_int8() noexcept {
+  return narrow_integer<int8_t>(get_int64());
 }
 simdjson_inline simdjson_result<double> document::get_double() noexcept {
   return get_root_value_iterator().get_root_double(true);
@@ -139177,9 +178008,36 @@ simdjson_inline simdjson_result<double> document::get_double() noexcept {
 simdjson_inline simdjson_result<double> document::get_double_in_string() noexcept {
   return get_root_value_iterator().get_root_double_in_string(true);
 }
+simdjson_inline simdjson_result<float> document::get_float() noexcept {
+  return get_root_value_iterator().get_root_float(true);
+}
+simdjson_inline simdjson_result<float> document::get_float_in_string() noexcept {
+  return get_root_value_iterator().get_root_float_in_string(true);
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> document::get_float32() noexcept {
+  float result;
+  SIMDJSON_TRY(get_float().get(result));
+  return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> document::get_float64() noexcept {
+  double result;
+  SIMDJSON_TRY(get_double().get(result));
+  return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<std::string_view> document::get_string(bool allow_replacement) noexcept {
   return get_root_value_iterator().get_root_string(true, allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> document::get_u8string(bool allow_replacement) noexcept {
+  std::string_view content;
+  SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+  return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code document::get_string(string_type& receiver, bool allow_replacement) noexcept {
   return get_root_value_iterator().get_root_string(receiver, true, allow_replacement);
@@ -139201,11 +178059,25 @@ template<> simdjson_inline simdjson_result<array> document::get() & noexcept { r
 template<> simdjson_inline simdjson_result<object> document::get() & noexcept { return get_object(); }
 template<> simdjson_inline simdjson_result<raw_json_string> document::get() & noexcept { return get_raw_json_string(); }
 template<> simdjson_inline simdjson_result<std::string_view> document::get() & noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> document::get() & noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_inline simdjson_result<double> document::get() & noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> document::get() & noexcept { return get_float(); }
 template<> simdjson_inline simdjson_result<uint64_t> document::get() & noexcept { return get_uint64(); }
 template<> simdjson_inline simdjson_result<int64_t> document::get() & noexcept { return get_int64(); }
 template<> simdjson_inline simdjson_result<uint32_t> document::get() & noexcept { return get_uint32(); }
 template<> simdjson_inline simdjson_result<int32_t> document::get() & noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> document::get() & noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> document::get() & noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> document::get() & noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> document::get() & noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> document::get() & noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> document::get() & noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<> simdjson_inline simdjson_result<bool> document::get() & noexcept { return get_bool(); }
 template<> simdjson_inline simdjson_result<value> document::get() & noexcept { return get_value(); }

@@ -139213,17 +178085,35 @@ template<> simdjson_warn_unused simdjson_inline error_code document::get(array&
 template<> simdjson_warn_unused simdjson_inline error_code document::get(object& out) & noexcept { return get_object().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(raw_json_string& out) & noexcept { return get_raw_json_string().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(std::string_view& out) & noexcept { return get_string(false).get(out); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::u8string_view& out) & noexcept { return get_u8string(false).get(out); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_warn_unused simdjson_inline error_code document::get(double& out) & noexcept { return get_double().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(float& out) & noexcept { return get_float().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(uint64_t& out) & noexcept { return get_uint64().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(int64_t& out) & noexcept { return get_int64().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(uint32_t& out) & noexcept { return get_uint32().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(int32_t& out) & noexcept { return get_int32().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(uint16_t& out) & noexcept { return get_uint16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(int16_t& out) & noexcept { return get_int16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(uint8_t& out) & noexcept { return get_uint8().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(int8_t& out) & noexcept { return get_int8().get(out); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::float32_t& out) & noexcept { return get_float32().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::float64_t& out) & noexcept { return get_float64().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<> simdjson_warn_unused simdjson_inline error_code document::get(bool& out) & noexcept { return get_bool().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(value& out) & noexcept { return get_value().get(out); }

 template<> simdjson_deprecated simdjson_inline simdjson_result<raw_json_string> document::get() && noexcept { return get_raw_json_string(); }
 template<> simdjson_deprecated simdjson_inline simdjson_result<std::string_view> document::get() && noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_deprecated simdjson_inline simdjson_result<std::u8string_view> document::get() && noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_deprecated simdjson_inline simdjson_result<double> document::get() && noexcept { return std::forward<document>(*this).get_double(); }
+template<> simdjson_deprecated simdjson_inline simdjson_result<float> document::get() && noexcept { return std::forward<document>(*this).get_float(); }
 template<> simdjson_deprecated simdjson_inline simdjson_result<uint64_t> document::get() && noexcept { return std::forward<document>(*this).get_uint64(); }
 template<> simdjson_deprecated simdjson_inline simdjson_result<int64_t> document::get() && noexcept { return std::forward<document>(*this).get_int64(); }
 template<> simdjson_deprecated simdjson_inline simdjson_result<bool> document::get() && noexcept { return std::forward<document>(*this).get_bool(); }
@@ -139562,6 +178452,22 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<westmere::ondemand::doc
   if (error()) { return error(); }
   return first.get_int32();
 }
+simdjson_inline simdjson_result<uint16_t> simdjson_result<westmere::ondemand::document>::get_uint16() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<westmere::ondemand::document>::get_int16() noexcept {
+  if (error()) { return error(); }
+  return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<westmere::ondemand::document>::get_uint8() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<westmere::ondemand::document>::get_int8() noexcept {
+  if (error()) { return error(); }
+  return first.get_int8();
+}
 simdjson_inline simdjson_result<double> simdjson_result<westmere::ondemand::document>::get_double() noexcept {
   if (error()) { return error(); }
   return first.get_double();
@@ -139570,10 +178476,36 @@ simdjson_inline simdjson_result<double> simdjson_result<westmere::ondemand::docu
   if (error()) { return error(); }
   return first.get_double_in_string();
 }
+simdjson_inline simdjson_result<float> simdjson_result<westmere::ondemand::document>::get_float() noexcept {
+  if (error()) { return error(); }
+  return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<westmere::ondemand::document>::get_float_in_string() noexcept {
+  if (error()) { return error(); }
+  return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<westmere::ondemand::document>::get_float32() noexcept {
+  if (error()) { return error(); }
+  return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<westmere::ondemand::document>::get_float64() noexcept {
+  if (error()) { return error(); }
+  return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<std::string_view> simdjson_result<westmere::ondemand::document>::get_string(bool allow_replacement) noexcept {
   if (error()) { return error(); }
   return first.get_string(allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<westmere::ondemand::document>::get_u8string(bool allow_replacement) noexcept {
+  if (error()) { return error(); }
+  return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code simdjson_result<westmere::ondemand::document>::get_string(string_type& receiver, bool allow_replacement) noexcept {
   if (error()) { return error(); }
@@ -139601,22 +178533,46 @@ simdjson_inline simdjson_result<bool> simdjson_result<westmere::ondemand::docume
 }

 template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<westmere::ondemand::document>::get() & noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<westmere::ondemand::document>::get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, westmere::ondemand::document>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>();
 }
 template<typename T>
-simdjson_deprecated simdjson_inline simdjson_result<T> simdjson_result<westmere::ondemand::document>::get() && noexcept {
+simdjson_deprecated simdjson_inline simdjson_result<T> simdjson_result<westmere::ondemand::document>::get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, westmere::ondemand::document>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return std::forward<westmere::ondemand::document>(first).get<T>();
 }
 template<typename T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<westmere::ondemand::document>::get(T &out) & noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<westmere::ondemand::document>::get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, westmere::ondemand::document>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>(out);
 }
 template<typename T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<westmere::ondemand::document>::get(T &out) && noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<westmere::ondemand::document>::get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, westmere::ondemand::document>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return std::forward<westmere::ondemand::document>(first).get<T>(out);
 }
@@ -139685,27 +178641,27 @@ simdjson_inline simdjson_result<westmere::ondemand::document>::operator westmere
 }
 simdjson_inline simdjson_result<westmere::ondemand::document>::operator uint64_t() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_uint64();
 }
 simdjson_inline simdjson_result<westmere::ondemand::document>::operator int64_t() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_int64();
 }
 simdjson_inline simdjson_result<westmere::ondemand::document>::operator double() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_double();
 }
 simdjson_inline simdjson_result<westmere::ondemand::document>::operator std::string_view() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_string();
 }
 simdjson_inline simdjson_result<westmere::ondemand::document>::operator westmere::ondemand::raw_json_string() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_raw_json_string();
 }
 simdjson_inline simdjson_result<westmere::ondemand::document>::operator bool() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_bool();
 }
 simdjson_inline simdjson_result<westmere::ondemand::document>::operator westmere::ondemand::value() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
@@ -139795,21 +178751,38 @@ simdjson_inline simdjson_result<uint64_t> document_reference::get_uint64() noexc
 simdjson_inline simdjson_result<uint64_t> document_reference::get_uint64_in_string() noexcept { return doc->get_root_value_iterator().get_root_uint64_in_string(false); }
 simdjson_inline simdjson_result<int64_t> document_reference::get_int64() noexcept { return doc->get_root_value_iterator().get_root_int64(false); }
 simdjson_inline simdjson_result<int64_t> document_reference::get_int64_in_string() noexcept { return doc->get_root_value_iterator().get_root_int64_in_string(false); }
-simdjson_inline simdjson_result<uint32_t> document_reference::get_uint32() noexcept {
-  uint64_t result;
-  SIMDJSON_TRY(doc->get_root_value_iterator().get_root_uint64(false).get(result));
-  if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<uint32_t>(result);
-}
-simdjson_inline simdjson_result<int32_t> document_reference::get_int32() noexcept {
-  int64_t result;
-  SIMDJSON_TRY(doc->get_root_value_iterator().get_root_int64(false).get(result));
-  if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<int32_t>(result);
-}
+simdjson_inline simdjson_result<uint32_t> document_reference::get_uint32() noexcept { return narrow_integer<uint32_t>(get_uint64()); }
+simdjson_inline simdjson_result<int32_t> document_reference::get_int32() noexcept { return narrow_integer<int32_t>(get_int64()); }
+simdjson_inline simdjson_result<uint16_t> document_reference::get_uint16() noexcept { return narrow_integer<uint16_t>(get_uint64()); }
+simdjson_inline simdjson_result<int16_t> document_reference::get_int16() noexcept { return narrow_integer<int16_t>(get_int64()); }
+simdjson_inline simdjson_result<uint8_t> document_reference::get_uint8() noexcept { return narrow_integer<uint8_t>(get_uint64()); }
+simdjson_inline simdjson_result<int8_t> document_reference::get_int8() noexcept { return narrow_integer<int8_t>(get_int64()); }
 simdjson_inline simdjson_result<double> document_reference::get_double() noexcept { return doc->get_root_value_iterator().get_root_double(false); }
-simdjson_inline simdjson_result<double> document_reference::get_double_in_string() noexcept { return doc->get_root_value_iterator().get_root_double(false); }
+simdjson_inline simdjson_result<double> document_reference::get_double_in_string() noexcept { return doc->get_root_value_iterator().get_root_double_in_string(false); }
+simdjson_inline simdjson_result<float> document_reference::get_float() noexcept { return doc->get_root_value_iterator().get_root_float(false); }
+simdjson_inline simdjson_result<float> document_reference::get_float_in_string() noexcept { return doc->get_root_value_iterator().get_root_float_in_string(false); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> document_reference::get_float32() noexcept {
+  float result;
+  SIMDJSON_TRY(get_float().get(result));
+  return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> document_reference::get_float64() noexcept {
+  double result;
+  SIMDJSON_TRY(get_double().get(result));
+  return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<std::string_view> document_reference::get_string(bool allow_replacement) noexcept { return doc->get_root_value_iterator().get_root_string(false, allow_replacement); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> document_reference::get_u8string(bool allow_replacement) noexcept {
+  std::string_view content;
+  SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+  return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code document_reference::get_string(string_type& receiver, bool allow_replacement) noexcept { return doc->get_root_value_iterator().get_root_string(receiver, false, allow_replacement); }
 simdjson_inline simdjson_result<std::string_view> document_reference::get_wobbly_string() noexcept { return doc->get_root_value_iterator().get_root_wobbly_string(false); }
@@ -139821,11 +178794,25 @@ template<> simdjson_inline simdjson_result<array> document_reference::get() & no
 template<> simdjson_inline simdjson_result<object> document_reference::get() & noexcept { return get_object(); }
 template<> simdjson_inline simdjson_result<raw_json_string> document_reference::get() & noexcept { return get_raw_json_string(); }
 template<> simdjson_inline simdjson_result<std::string_view> document_reference::get() & noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> document_reference::get() & noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_inline simdjson_result<double> document_reference::get() & noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> document_reference::get() & noexcept { return get_float(); }
 template<> simdjson_inline simdjson_result<uint64_t> document_reference::get() & noexcept { return get_uint64(); }
 template<> simdjson_inline simdjson_result<int64_t> document_reference::get() & noexcept { return get_int64(); }
 template<> simdjson_inline simdjson_result<uint32_t> document_reference::get() & noexcept { return get_uint32(); }
 template<> simdjson_inline simdjson_result<int32_t> document_reference::get() & noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> document_reference::get() & noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> document_reference::get() & noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> document_reference::get() & noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> document_reference::get() & noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> document_reference::get() & noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> document_reference::get() & noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<> simdjson_inline simdjson_result<bool> document_reference::get() & noexcept { return get_bool(); }
 template<> simdjson_inline simdjson_result<value> document_reference::get() & noexcept { return get_value(); }
 #if SIMDJSON_EXCEPTIONS
@@ -139971,6 +178958,22 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<westmere::ondemand::doc
   if (error()) { return error(); }
   return first.get_int32();
 }
+simdjson_inline simdjson_result<uint16_t> simdjson_result<westmere::ondemand::document_reference>::get_uint16() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<westmere::ondemand::document_reference>::get_int16() noexcept {
+  if (error()) { return error(); }
+  return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<westmere::ondemand::document_reference>::get_uint8() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<westmere::ondemand::document_reference>::get_int8() noexcept {
+  if (error()) { return error(); }
+  return first.get_int8();
+}
 simdjson_inline simdjson_result<double> simdjson_result<westmere::ondemand::document_reference>::get_double() noexcept {
   if (error()) { return error(); }
   return first.get_double();
@@ -139979,10 +178982,36 @@ simdjson_inline simdjson_result<double> simdjson_result<westmere::ondemand::docu
   if (error()) { return error(); }
   return first.get_double_in_string();
 }
+simdjson_inline simdjson_result<float> simdjson_result<westmere::ondemand::document_reference>::get_float() noexcept {
+  if (error()) { return error(); }
+  return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<westmere::ondemand::document_reference>::get_float_in_string() noexcept {
+  if (error()) { return error(); }
+  return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<westmere::ondemand::document_reference>::get_float32() noexcept {
+  if (error()) { return error(); }
+  return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<westmere::ondemand::document_reference>::get_float64() noexcept {
+  if (error()) { return error(); }
+  return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<std::string_view> simdjson_result<westmere::ondemand::document_reference>::get_string(bool allow_replacement) noexcept {
   if (error()) { return error(); }
   return first.get_string(allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<westmere::ondemand::document_reference>::get_u8string(bool allow_replacement) noexcept {
+  if (error()) { return error(); }
+  return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code simdjson_result<westmere::ondemand::document_reference>::get_string(string_type& receiver, bool allow_replacement) noexcept {
   if (error()) { return error(); }
@@ -140009,22 +179038,46 @@ simdjson_inline simdjson_result<bool> simdjson_result<westmere::ondemand::docume
   return first.is_null();
 }
 template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<westmere::ondemand::document_reference>::get() & noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<westmere::ondemand::document_reference>::get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, westmere::ondemand::document_reference>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>();
 }
 template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<westmere::ondemand::document_reference>::get() && noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<westmere::ondemand::document_reference>::get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, westmere::ondemand::document_reference>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return std::forward<westmere::ondemand::document_reference>(first).get<T>();
 }
 template <class T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<westmere::ondemand::document_reference>::get(T &out) & noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<westmere::ondemand::document_reference>::get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, westmere::ondemand::document_reference>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>(out);
 }
 template <class T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<westmere::ondemand::document_reference>::get(T &out) && noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<westmere::ondemand::document_reference>::get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, westmere::ondemand::document_reference>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return std::forward<westmere::ondemand::document_reference>(first).get<T>(out);
 }
@@ -140086,27 +179139,27 @@ simdjson_inline simdjson_result<westmere::ondemand::document_reference>::operato
 }
 simdjson_inline simdjson_result<westmere::ondemand::document_reference>::operator uint64_t() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_uint64();
 }
 simdjson_inline simdjson_result<westmere::ondemand::document_reference>::operator int64_t() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_int64();
 }
 simdjson_inline simdjson_result<westmere::ondemand::document_reference>::operator double() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_double();
 }
 simdjson_inline simdjson_result<westmere::ondemand::document_reference>::operator std::string_view() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_string();
 }
 simdjson_inline simdjson_result<westmere::ondemand::document_reference>::operator westmere::ondemand::raw_json_string() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_raw_json_string();
 }
 simdjson_inline simdjson_result<westmere::ondemand::document_reference>::operator bool() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_bool();
 }
 simdjson_inline simdjson_result<westmere::ondemand::document_reference>::operator westmere::ondemand::value() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
@@ -140172,6 +179225,7 @@ simdjson_warn_unused simdjson_inline error_code simdjson_result<westmere::ondema
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */

 #include <algorithm>
+#include <cstring>
 #include <stdexcept>

 namespace simdjson {
@@ -140258,23 +179312,20 @@ simdjson_inline document_stream::document_stream(
   const uint8_t *_buf,
   size_t _len,
   size_t _batch_size,
-  bool _allow_comma_separated
+  bool _allow_comma_separated,
+  stream_format _format
 ) noexcept
   : parser{&_parser},
     buf{_buf},
     len{_len},
     batch_size{_batch_size <= MINIMAL_BATCH_SIZE ? MINIMAL_BATCH_SIZE : _batch_size},
     allow_comma_separated{_allow_comma_separated},
+    format{_format},
     error{SUCCESS}
     #ifdef SIMDJSON_THREADS_ENABLED
     , use_thread(_parser.threaded) // we need to make a copy because _parser.threaded can change
     #endif
 {
-#ifdef SIMDJSON_THREADS_ENABLED
-  if(worker.get() == nullptr) {
-    error = MEMALLOC;
-  }
-#endif
 }

 simdjson_inline document_stream::document_stream() noexcept
@@ -140283,6 +179334,7 @@ simdjson_inline document_stream::document_stream() noexcept
     len{0},
     batch_size{0},
     allow_comma_separated{false},
+    format{stream_format::whitespace_delimited},
     error{UNINITIALIZED}
     #ifdef SIMDJSON_THREADS_ENABLED
     , use_thread(false)
@@ -140302,6 +179354,9 @@ inline size_t document_stream::size_in_bytes() const noexcept {
 }

 inline size_t document_stream::truncated_bytes() const noexcept {
+  // Stage 1 returns EMPTY on zero-length input before it writes the index
+  // sentinels read below, so they would still hold a previous stream's values.
+  if (len == 0) { return 0; }
   if(error == CAPACITY) { return len - batch_start; }
   return parser->implementation->structural_indexes[parser->implementation->n_structural_indexes] - parser->implementation->structural_indexes[parser->implementation->n_structural_indexes + 1];
 }
@@ -140382,13 +179437,20 @@ inline void document_stream::start() noexcept {
     error = run_stage1(*parser, batch_start);
   }
   if (error) { return; }
-  doc_index = batch_start;
+  // For json_sequence mode, structural_indexes[0] points to the actual JSON value
+  // after the RS delimiter and any following whitespace. For regular mode, it is
+  // the offset from batch_start to the first document in the batch.
+  doc_index = batch_start + parser->implementation->structural_indexes[0];
   doc = document(json_iterator(&buf[batch_start], parser));
   doc.iter._streaming = true;

   #ifdef SIMDJSON_THREADS_ENABLED
   if (use_thread && next_batch_start() < len) {
     // Kick off the first thread on next batch if needed
+    if (worker.get() == nullptr) {
+      worker.reset(new(std::nothrow) stage1_worker());
+      if (worker.get() == nullptr) { error = MEMALLOC; return; }
+    }
     error = stage1_thread_parser.allocate(batch_size);
     if (error) { return; }
     worker->start_thread();
@@ -140463,12 +179525,69 @@ inline void document_stream::next() noexcept {
        */

       if (error) { continue; } // If the error was EMPTY, we may want to load another batch.
-      doc_index = batch_start;
+      doc_index = batch_start + parser->implementation->structural_indexes[0];
     }
   }
 }

+simdjson_inline uint8_t document_stream::document_delimiter() const noexcept {
+  switch (format) {
+    case stream_format::newline_delimited: return '\n';
+    case stream_format::json_sequence: return 0x1E;
+    default: return 0;
+  }
+}
+
+simdjson_inline bool document_stream::skip_to_delimiter(uint8_t delimiter) noexcept {
+  const uint8_t *const base = &buf[batch_start];
+  const token_position pos = doc.iter.position();
+  const token_position end = doc.iter.end_position();
+  if (pos >= end) { return false; }
+  const size_t here = size_t(doc.iter.token.peek(pos) - base);
+  const size_t batch_len =
+      (len - batch_start < batch_size) ? len - batch_start : batch_size;
+  if (here >= batch_len) { return false; }
+  const uint8_t *const found = static_cast<const uint8_t *>(
+      std::memchr(base + here, delimiter, batch_len - here));
+  if (found == nullptr) { return false; }
+
+  const uint32_t boundary = uint32_t(found - base);
+  // The answer is near `pos`: the delimiter ends the current document, while
+  // `end` spans the whole batch. Gallop first so the cost follows the distance
+  // rather than the size of the batch.
+  token_position lo = pos;
+  size_t hop = 1;
+  while (lo + hop < end && lo[hop] < boundary) { lo += hop; hop <<= 1; }
+  token_position hi = (lo + hop < end) ? lo + hop : end;
+  while (lo < hi) {
+    const token_position mid = lo + ((hi - lo) >> 1);
+    if (*mid < boundary) { lo = mid + 1; } else { hi = mid; }
+  }
+  doc.iter.token.set_position(lo);
+  return true;
+}
+
 inline void document_stream::next_document() noexcept {
+  // A delimiter that cannot occur inside a document tells us where the current
+  // one ends, so we can jump there instead of walking every structural. Only
+  // valid while the iterator is still inside the document: a consumed document
+  // already sits on the next one's first token, and skip_child() returns at
+  // once for it.
+  //
+  // The jump does not structure-validate the unread remainder of the current
+  // document: under newline_delimited / json_sequence the next delimiter is
+  // assumed to be the true document boundary. Callers that leave depth() > 0
+  // while violating that contract (e.g. pretty multi-line JSON under
+  // newline_delimited) can mis-align following documents; use
+  // whitespace_delimited if unsure.
+  const uint8_t delimiter = document_delimiter();
+  if (delimiter != 0 && !error && doc.iter.depth() > 0 &&
+      skip_to_delimiter(delimiter)) {
+    doc.iter._depth = 1;
+    doc.iter._string_buf_loc = parser->string_buf.get();
+    doc.iter._root = doc.iter.position();
+    return;
+  }
   // Go to next place where depth=0 (document depth)
   error = doc.iter.skip_child(0);
   if (error) { return; }
@@ -140492,10 +179611,35 @@ inline error_code document_stream::run_stage1(ondemand::parser &p, size_t _batch
   // This code only updates the structural index in the parser, it does not update any json_iterator
   // instance.
   size_t remaining = len - _batch_start;
+  stage1_mode mode;
   if (remaining <= batch_size) {
-    return p.implementation->stage1(&buf[_batch_start], remaining, stage1_mode::streaming_final);
+    // Final batch
+    switch (format) {
+      case stream_format::json_sequence:
+        mode = stage1_mode::json_sequence_final;
+        break;
+      case stream_format::comma_delimited:
+        mode = stage1_mode::comma_delimited_final;
+        break;
+      default:
+        mode = stage1_mode::streaming_final;
+        break;
+    }
+    return p.implementation->stage1(&buf[_batch_start], remaining, mode);
   } else {
-    return p.implementation->stage1(&buf[_batch_start], batch_size, stage1_mode::streaming_partial);
+    // Partial batch
+    switch (format) {
+      case stream_format::json_sequence:
+        mode = stage1_mode::json_sequence_partial;
+        break;
+      case stream_format::comma_delimited:
+        mode = stage1_mode::comma_delimited_partial;
+        break;
+      default:
+        mode = stage1_mode::streaming_partial;
+        break;
+    }
+    return p.implementation->stage1(&buf[_batch_start], batch_size, mode);
   }
 }

@@ -140504,11 +179648,19 @@ simdjson_inline size_t document_stream::iterator::current_index() const noexcept
 }

 simdjson_inline std::string_view document_stream::iterator::source() const noexcept {
-  auto depth = stream->doc.iter.depth();
+  // On error (e.g., CAPACITY), there is no document to walk: return the rest of
+  // the input, as the DOM document_stream does.
+  if (stream->error) {
+    return std::string_view(reinterpret_cast<const char*>(stream->buf) + current_index(), stream->len - current_index());
+  }
+  // Always walk from the root of the document, whatever the current position
+  // of the document iterator: the user may have already consumed part of the
+  // document, so the iterator's current depth must not be used here.
+  depth_t depth = 1;
   auto cur_struct_index = stream->doc.iter._root - stream->parser->implementation->structural_indexes.get();

-  // If at root, process the first token to determine if scalar value
-  if (stream->doc.iter.at_root()) {
+  // Process the first token to determine if scalar value
+  {
     switch (stream->buf[stream->batch_start + stream->parser->implementation->structural_indexes[cur_struct_index]]) {
       case '{': case '[':   // Depth=1 already at start of document
         break;
@@ -140516,14 +179668,47 @@ simdjson_inline std::string_view document_stream::iterator::source() const noexc
         depth--;
         break;
       default:    // Scalar value document
-        // TODO: We could remove trailing whitespaces
         // This returns a string spanning from start of value to the beginning of the next document (excluded)
         {
           auto next_index = stream->batch_start + stream->parser->implementation->structural_indexes[++cur_struct_index];
           // normally the length would be next_index - current_index() - 1, except for the last document
           size_t svlen = next_index - current_index();
           const char *start = reinterpret_cast<const char*>(stream->buf) + current_index();
-          while(svlen > 1 && (std::isspace(start[svlen-1]) || start[svlen-1] == '\0')) {
+          // When the scalar is followed by a truncated document, the structural
+          // indexes of that document were dropped and next_index is the end of
+          // the input, so we bound the scalar by scanning the token itself.
+          size_t token_len = 0;
+          if (*start == '"') {
+            token_len = 1;
+            while (token_len < svlen) {
+              char c = start[token_len++];
+              if (c == '\\') {
+                token_len++;
+              } else if (c == '"') {
+                break;
+              }
+            }
+          } else {
+            while (token_len < svlen) {
+              char c = start[token_len];
+              if (std::isspace(static_cast<unsigned char>(c)) || c == ',' || c == '{' || c == '[' || c == '\0' || static_cast<uint8_t>(c) == 0x1E) {
+                break;
+              }
+              token_len++;
+            }
+          }
+          if (token_len > 0 && token_len < svlen) {
+            svlen = token_len;
+          }
+          // Trim trailing whitespace, NUL, and RS (0x1E). In RFC 7464
+          // json_sequence mode the scanner classifies RS as a scalar
+          // character, so an RS-prefixed scalar document (number / true /
+          // false / null / string) has no closing structural index and the
+          // slice runs all the way up to the next document's RS. RS cannot
+          // legally appear in a JSON value at the source level (control
+          // characters in strings must be escaped as \u001E), so stripping
+          // it is safe in every stream_format.
+          while(svlen > 1 && (std::isspace(static_cast<unsigned char>(start[svlen-1])) || start[svlen-1] == '\0' || static_cast<uint8_t>(start[svlen-1]) == 0x1E || (stream->format == stream_format::comma_delimited && start[svlen-1] == ','))) {
             svlen--;
           }
           return std::string_view(start, svlen);
@@ -140648,11 +179833,19 @@ simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> field::un
   return answer;
 }

+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_warn_unused simdjson_result<std::u8string_view> field::unescaped_u8key(bool allow_replacement) noexcept {
+  std::string_view key;
+  SIMDJSON_TRY( unescaped_key(allow_replacement).get(key) );
+  return internal::as_u8string_view(key);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 template <typename string_type>
 simdjson_inline simdjson_warn_unused error_code field::unescaped_key(string_type& receiver, bool allow_replacement) noexcept {
   std::string_view key;
   SIMDJSON_TRY( unescaped_key(allow_replacement).get(key) );
-  receiver = key;
+  internal::assign_utf8(receiver, key);
   return SUCCESS;
 }

@@ -140674,6 +179867,12 @@ simdjson_inline std::string_view field::escaped_key() const noexcept {
   return std::string_view(reinterpret_cast<const char*>(first.buf), end_quote - first.buf);
 }

+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline std::u8string_view field::escaped_u8key() const noexcept {
+  return internal::as_u8string_view(escaped_key());
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 simdjson_inline value &field::value() & noexcept {
   return second;
 }
@@ -140718,11 +179917,25 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<westmere::onde
   return first.escaped_key();
 }

+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<westmere::ondemand::field>::escaped_u8key() noexcept {
+  if (error()) { return error(); }
+  return first.escaped_u8key();
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 simdjson_inline simdjson_result<std::string_view> simdjson_result<westmere::ondemand::field>::unescaped_key(bool allow_replacement) noexcept {
   if (error()) { return error(); }
   return first.unescaped_key(allow_replacement);
 }

+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<westmere::ondemand::field>::unescaped_u8key(bool allow_replacement) noexcept {
+  if (error()) { return error(); }
+  return first.unescaped_u8key(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 template<typename string_type>
 simdjson_warn_unused simdjson_inline error_code simdjson_result<westmere::ondemand::field>::unescaped_key(string_type &receiver, bool allow_replacement) noexcept {
   if (error()) { return error(); }
@@ -140766,6 +179979,9 @@ simdjson_inline json_iterator::json_iterator(json_iterator &&other) noexcept
     _depth{other._depth},
     _root{other._root},
     _streaming{other._streaming}
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+    , _allow_incomplete_json{other._allow_incomplete_json}
+#endif
 {
   other.parser = nullptr;
 }
@@ -140777,6 +179993,9 @@ simdjson_inline json_iterator &json_iterator::operator=(json_iterator &&other) n
   _depth = other._depth;
   _root = other._root;
   _streaming = other._streaming;
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  _allow_incomplete_json = other._allow_incomplete_json;
+#endif
   other.parser = nullptr;
   return *this;
 }
@@ -140803,7 +180022,8 @@ simdjson_inline json_iterator::json_iterator(const uint8_t *buf, ondemand::parse
       _string_buf_loc{parser->string_buf.get()},
       _depth{1},
       _root{parser->implementation->structural_indexes.get()},
-      _streaming{streaming}
+      _streaming{streaming},
+      _allow_incomplete_json{true}

 {
   logger::log_headers();
@@ -140875,7 +180095,8 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::skip_child(depth_
 #endif // SIMDJSON_CHECK_EOF
       break;
     case '"':
-      if(*peek() == ':') {
+      // At the end, peek() would read the sentinel, which points into the padding.
+      if(!at_end() && *peek() == ':') {
         // We are at a key!!!
         // This might happen if you just started an object and you skip it immediately.
         // Performance note: it would be nice to get rid of this check as it is somewhat
@@ -140918,7 +180139,7 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::skip_child(depth_
     }
   }

-  return report_error(TAPE_ERROR, "not enough close braces");
+  return report_error(INCOMPLETE_ARRAY_OR_OBJECT, "not enough close braces");
 }

 SIMDJSON_POP_DISABLE_WARNINGS
@@ -140935,6 +180156,17 @@ simdjson_inline bool json_iterator::streaming() const noexcept {
   return _streaming;
 }

+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+simdjson_inline bool json_iterator::allow_incomplete_json() const noexcept {
+  return _allow_incomplete_json;
+}
+
+simdjson_inline size_t json_iterator::remaining_input_length(const uint8_t *json) const noexcept {
+  const uint8_t *end = token.buf + parser->_document_len;
+  return json < end ? size_t(end - json) : 0;
+}
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+
 simdjson_inline token_position json_iterator::root_position() const noexcept {
   return _root;
 }
@@ -141217,7 +180449,7 @@ inline std::ostream& operator<<(std::ostream& out, json_type type) noexcept {
         case json_type::string: out << "string"; break;
         case json_type::boolean: out << "boolean"; break;
         case json_type::null: out << "null"; break;
-        default: SIMDJSON_UNREACHABLE();
+        case json_type::unknown: out << "unknown"; break;
     }
     return out;
 }
@@ -141556,6 +180788,10 @@ inline void log_line(const json_iterator &iter, token_position index, depth_t de
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_iterator.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value-inl.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/jsonpathutil.h" */
+/* amalgamation skipped (editor-only): #include <utility> */
+/* amalgamation skipped (editor-only): #if SIMDJSON_SUPPORTS_CONCEPTS */
+/* amalgamation skipped (editor-only): #include <tuple> // std::forward_as_tuple/get for the variadic for_each adapter */
+/* amalgamation skipped (editor-only): #endif */
 /* amalgamation skipped (editor-only): #if SIMDJSON_STATIC_REFLECTION */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_string_builder.h"  // for constevalutil::fixed_string */
 /* amalgamation skipped (editor-only): #include <meta> */
@@ -141585,12 +180821,21 @@ simdjson_inline simdjson_result<value> object::find_field_unordered(const std::s
   return value(iter.child());
 }
 simdjson_inline simdjson_result<value> object::operator[](const std::string_view key) & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   return find_field_unordered(key);
 }
 simdjson_inline simdjson_result<value> object::operator[](const std::string_view key) && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   return std::forward<object>(*this).find_field_unordered(key);
 }
 simdjson_inline simdjson_result<value> object::find_field(const std::string_view key) & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   bool has_value;
   SIMDJSON_TRY( iter.find_field_raw(key).get(has_value) );
   if (!has_value) {
@@ -141600,6 +180845,9 @@ simdjson_inline simdjson_result<value> object::find_field(const std::string_view
   return value(iter.child());
 }
 simdjson_inline simdjson_result<value> object::find_field(const std::string_view key) && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   bool has_value;
   SIMDJSON_TRY( iter.find_field_raw(key).get(has_value) );
   if (!has_value) {
@@ -141609,6 +180857,150 @@ simdjson_inline simdjson_result<value> object::find_field(const std::string_view
   return value(iter.child());
 }

+#if SIMDJSON_SUPPORTS_CONCEPTS
+template <typename Selector, typename Func>
+  requires key_selector_type<Selector> &&
+           std::is_invocable_v<Func&, std::size_t, value>
+simdjson_flatten simdjson_inline for_each_result object::for_each(Func&& on_match)
+    noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, value>) {
+  // Single pass driven directly by the value_iterator, mirroring
+  // find_field_unordered_raw + value(iter.child()). Compared to walking via
+  // object_iterator/field, this avoids constructing a simdjson_result<field> and
+  // a field (key + value) for every field -- and the development-check bookkeeping
+  // in object_iterator -- building a value only for the fields that actually match.
+  // We operate on a copy of the iterator, as object::begin() would.
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  // Mirror object::begin(): for_each must start at the beginning of the object,
+  // not from some position left behind by a prior find_field on the same object.
+  if (!iter.is_at_iterator_start()) { return {OUT_OF_ORDER_ITERATION, 0}; }
+#endif
+  value_iterator it = iter;
+  std::size_t matched = 0;
+  // Track which selector indices have already matched, as a compile-time bitset
+  // (one 64-bit word per 64 keys): the callback fires once per key, on its first
+  // occurrence, and we stop as soon as every key has matched.
+  constexpr std::size_t seen_words = (Selector::size() + 63) / 64;
+  std::array<std::uint64_t, seen_words> seen{};
+  while (it.is_open()) {
+    raw_json_string key;
+    error_code error;
+    std::size_t idx;
+    if constexpr (Selector::window.ok) {
+      // A window selector confirms a key from its raw bytes alone (the closing
+      // quote bounds it), so we take the length-free path: field_key (no backward
+      // scan, mirroring the ordered find_field) feeding match_raw(raw_json_string).
+      if ((error = it.field_key().get(key))) { it.abandon(); return {error, matched}; }
+      if ((error = it.field_value())) { it.abandon(); return {error, matched}; }
+      idx = Selector::match_raw(key);
+    } else {
+      // Otherwise derive the key length from the structural index (the following
+      // ':' token) to feed the length-aware hash, avoiding a forward SIMD scan.
+      std::size_t key_len;
+      if ((error = it.field_key_with_length(key, key_len))) { it.abandon(); return {error, matched}; }
+      if ((error = it.field_value())) { it.abandon(); return {error, matched}; }
+      idx = Selector::match_raw(key.raw(), key_len);
+    }
+    if (idx < Selector::size()) {
+      const std::uint64_t seen_bit = std::uint64_t{1} << (idx & 63);
+      std::uint64_t &seen_word = seen[idx >> 6];
+      if (!(seen_word & seen_bit)) {
+        seen_word |= seen_bit;
+        value matched_value(it.child());
+        // The callback may return void or anything convertible to error_code
+        // (error_code itself, or a for_each_result from a nested for_each). When
+        // it yields an error_code, we stop at the first non-SUCCESS result and
+        // propagate it so the caller can surface value-parse errors (for example,
+        // a type mismatch on a matched field). A void-returning callback is
+        // responsible for handling its own errors.
+        if constexpr (std::is_convertible_v<decltype(on_match(idx, matched_value)), error_code>) {
+          // Unlike the internal-error paths above, a callback error does not
+          // abandon the iterator: we leave it recoverable so the caller can keep
+          // using the object (or its parent) after handling the error.
+          if ((error = on_match(idx, matched_value))) { return {error, matched}; }
+        } else {
+          on_match(idx, matched_value);
+        }
+        if (++matched >= Selector::size()) { break; }
+      }
+    }
+    // Mirror object_iterator::operator++'s safety rail: if the callback consumed
+    // the value and left the iterator closed or in error (e.g. a void callback
+    // that swallowed a fatal sub-iteration error), stop here rather than calling
+    // skip_child on a closed iterator.
+    if (!it.is_open()) { break; }
+    // Skip the value (a no-op if the callback consumed it) and step to the next
+    // field; has_next_field() ends the container on '}', which closes the loop.
+    if ((error = it.skip_child())) { it.abandon(); return {error, matched}; }
+    if ((error = it.has_next_field().error())) { return {error, matched}; }
+  }
+  return {SUCCESS, matched};
+}
+
+namespace key_selector_for_each_detail {
+
+// Dispatch a matched value to the I-th handler in the tuple (0-based). A handler
+// is either an invocable taking the value (void- or error_code-returning) or a
+// deserialization target, in which case we assign via value::get. We always
+// return error_code so the core (index, value) for_each can treat the adapter
+// uniformly.
+template <typename Tuple, std::size_t... Is>
+simdjson_really_inline error_code dispatch_value(
+    std::size_t idx, Tuple& handlers, value v, std::index_sequence<Is...>) {
+  error_code err = SUCCESS;
+  auto try_one = [&](auto Ic) {
+    constexpr std::size_t I = decltype(Ic)::value;
+    if (idx == I) {
+      auto&& h = std::get<I>(handlers);
+      using H = std::remove_reference_t<decltype(h)>;
+      if constexpr (std::is_invocable_v<H&, value>) {
+        // A handler returning void runs for its side effects; one returning
+        // anything convertible to error_code (error_code, or a for_each_result
+        // from a nested for_each) has its error captured and propagated.
+        if constexpr (std::is_convertible_v<decltype(h(v)), error_code>) {
+          err = h(v);
+        } else {
+          h(v);
+        }
+      } else {
+        // Direct deserialization target: assign the matched value into it.
+        err = v.get(h);
+      }
+    }
+  };
+  (try_one(std::integral_constant<std::size_t, Is>{}), ...);
+  return err;
+}
+
+} // namespace key_selector_for_each_detail
+
+template <typename Selector, typename... Handlers>
+  requires key_selector_type<Selector> &&
+           (sizeof...(Handlers) == Selector::size()) &&
+           (key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_flatten simdjson_inline for_each_result object::for_each(Handlers&&... on_match)
+    noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+  auto handlers = std::forward_as_tuple(std::forward<Handlers>(on_match)...);
+  // Reuse the single (index, value) implementation via a tiny adapter.
+  // The adapter is called once per *matched* key (very few); the hot path
+  // (iteration + match_raw + seen bitset) stays exactly the same.
+  return this->template for_each<Selector>(
+      [&](std::size_t i, value v) -> error_code {
+        return key_selector_for_each_detail::dispatch_value(
+            i, handlers, v, std::make_index_sequence<Selector::size()>{});
+      });
+}
+
+template <constevalutil::fixed_string... Keys, typename... Handlers>
+  requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+           (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+           (key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline for_each_result object::for_each(Handlers&&... on_match)
+    noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+  using Selector = key_selector<Keys...>;
+  return this->template for_each<Selector>(std::forward<Handlers>(on_match)...);
+}
+#endif
+
 simdjson_inline simdjson_result<object> object::start(value_iterator &iter) noexcept {
   SIMDJSON_TRY( iter.start_object().error() );
   return object(iter);
@@ -141644,6 +181036,9 @@ simdjson_warn_unused simdjson_inline error_code object::consume() noexcept {
 }

 simdjson_inline simdjson_result<std::string_view> object::raw_json() noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   const uint8_t * starting_point{iter.peek_start()};
   auto error = consume();
   if(error) { return error; }
@@ -141665,9 +181060,17 @@ simdjson_inline object::object(const value_iterator &_iter) noexcept
 {
 }

-simdjson_inline simdjson_result<object_iterator> object::begin() noexcept {
+simdjson_inline simdjson_result<object_iterator> object::begin() & noexcept {
 #if SIMDJSON_DEVELOPMENT_CHECKS
-  if (!iter.is_at_iterator_start()) { return OUT_OF_ORDER_ITERATION; }
+  if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
+  return object_iterator(iter, this);
+#endif
+  return object_iterator(iter);
+}
+simdjson_inline simdjson_result<object_iterator> object::begin() && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  // The object is a temporary that the iterator may outlive: do not lock it.
+  if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
 #endif
   return object_iterator(iter);
 }
@@ -141676,7 +181079,9 @@ simdjson_inline simdjson_result<object_iterator> object::end() noexcept {
 }

 inline simdjson_result<value> object::at_pointer(std::string_view json_pointer) noexcept {
-  if (json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
+  // An empty pointer has no json_pointer[0]: with a default-constructed
+  // std::string_view, reading it dereferences a null pointer.
+  if (json_pointer.empty() || json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
   json_pointer = json_pointer.substr(1);
   size_t slash = json_pointer.find('/');
   std::string_view key = json_pointer.substr(0, slash);
@@ -141778,6 +181183,9 @@ simdjson_inline simdjson_result<size_t> object::count_fields() & noexcept {
 }

 simdjson_inline simdjson_result<bool> object::is_empty() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   bool is_not_empty;
   auto error = iter.reset_object().get(is_not_empty);
   if(error) { return error; }
@@ -141785,9 +181193,41 @@ simdjson_inline simdjson_result<bool> object::is_empty() & noexcept {
 }

 simdjson_inline simdjson_result<bool> object::reset() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   return iter.reset_object();
 }

+simdjson_inline object_position object::get_current_position() const noexcept {
+  return object_position{iter.position(), iter.json_iter().depth()};
+}
+
+simdjson_inline error_code object::revert_position(object_position position) noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
+  // json_iterator::reenter_child() requires the live depth to be exactly
+  // one level shallower than the target (matching how every other depth
+  // transition in this iterator works), and under SIMDJSON_DEVELOPMENT_CHECKS
+  // additionally validates against the parser's per-depth container-start
+  // bookkeeping. Neither applies here: depending on what was captured and
+  // what has happened since (a scalar field fully consumed, a compound
+  // value left open, a find_field() miss that scanned past everything),
+  // the live depth when reverting can be any number of levels away from
+  // the captured one, and the captured depth is not necessarily a
+  // container's own start. reenter_at() moves directly, matching how
+  // reset_object() itself repositions without going through reenter_child().
+  iter.reenter_at(position.position, position.depth);
+  return SUCCESS;
+}
+
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline void object::set_locked(bool _locked) noexcept {
+  locked = _locked;
+}
+#endif
+
 #if SIMDJSON_SUPPORTS_CONCEPTS && SIMDJSON_STATIC_REFLECTION

 template<constevalutil::fixed_string... FieldNames, typename T>
@@ -141845,10 +181285,14 @@ simdjson_inline simdjson_result<westmere::ondemand::object>::simdjson_result(wes
 simdjson_inline simdjson_result<westmere::ondemand::object>::simdjson_result(error_code error) noexcept
     : implementation_simdjson_result_base<westmere::ondemand::object>(error) {}

-simdjson_inline simdjson_result<westmere::ondemand::object_iterator> simdjson_result<westmere::ondemand::object>::begin() noexcept {
+simdjson_inline simdjson_result<westmere::ondemand::object_iterator> simdjson_result<westmere::ondemand::object>::begin() & noexcept {
   if (error()) { return error(); }
   return first.begin();
 }
+simdjson_inline simdjson_result<westmere::ondemand::object_iterator> simdjson_result<westmere::ondemand::object>::begin() && noexcept {
+  if (error()) { return error(); }
+  return std::move(first).begin();
+}
 simdjson_inline simdjson_result<westmere::ondemand::object_iterator> simdjson_result<westmere::ondemand::object>::end() noexcept {
   if (error()) { return error(); }
   return first.end();
@@ -141902,11 +181346,55 @@ simdjson_inline error_code simdjson_result<westmere::ondemand::object>::for_each
   return first.for_each_at_path_with_wildcard(json_path, std::forward<Func>(callback));
 }

+#if SIMDJSON_SUPPORTS_CONCEPTS
+template <typename Selector, typename Func>
+  requires westmere::ondemand::key_selector_type<Selector> &&
+           std::is_invocable_v<Func&, std::size_t, westmere::ondemand::value>
+simdjson_inline westmere::ondemand::for_each_result
+simdjson_result<westmere::ondemand::object>::for_each(Func&& on_match)
+    noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, westmere::ondemand::value>) {
+  if (error()) { return {error(), 0}; }
+  return first.template for_each<Selector>(std::forward<Func>(on_match));
+}
+
+template <typename Selector, typename... Handlers>
+  requires westmere::ondemand::key_selector_type<Selector> &&
+           (sizeof...(Handlers) == Selector::size()) &&
+           (westmere::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline westmere::ondemand::for_each_result
+simdjson_result<westmere::ondemand::object>::for_each(Handlers&&... on_match)
+    noexcept(westmere::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+  if (error()) { return {error(), 0}; }
+  return first.template for_each<Selector>(std::forward<Handlers>(on_match)...);
+}
+
+template <constevalutil::fixed_string... Keys, typename... Handlers>
+  requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+           (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+           (westmere::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline westmere::ondemand::for_each_result
+simdjson_result<westmere::ondemand::object>::for_each(Handlers&&... on_match)
+    noexcept(westmere::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+  if (error()) { return {error(), 0}; }
+  return first.template for_each<Keys...>(std::forward<Handlers>(on_match)...);
+}
+#endif
+
 inline simdjson_result<bool> simdjson_result<westmere::ondemand::object>::reset() noexcept {
   if (error()) { return error(); }
   return first.reset();
 }

+inline simdjson_result<westmere::ondemand::object_position> simdjson_result<westmere::ondemand::object>::get_current_position() noexcept {
+  if (error()) { return error(); }
+  return first.get_current_position();
+}
+
+inline error_code simdjson_result<westmere::ondemand::object>::revert_position(westmere::ondemand::object_position position) noexcept {
+  if (error()) { return error(); }
+  return first.revert_position(position);
+}
+
 inline simdjson_result<bool> simdjson_result<westmere::ondemand::object>::is_empty() noexcept {
   if (error()) { return error(); }
   return first.is_empty();
@@ -141950,6 +181438,61 @@ simdjson_inline object_iterator::object_iterator(const value_iterator &_iter) no
   : iter{_iter}
 {}

+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline object_iterator::object_iterator(const value_iterator &_iter, object* _parent) noexcept
+  : parent{_parent}, iter{_iter}
+{
+  if (parent) parent->set_locked(true);
+}
+#endif
+
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline object_iterator::~object_iterator() noexcept
+{
+  if (parent) parent->set_locked(false);
+}
+
+simdjson_inline object_iterator::object_iterator(object_iterator&& other) noexcept
+  : has_been_referenced{other.has_been_referenced},
+    parent{other.parent},
+    iter{std::move(other.iter)}
+{
+  other.parent = nullptr;
+}
+
+simdjson_inline object_iterator& object_iterator::operator=(object_iterator&& other) noexcept {
+  if (this != &other)
+  {
+    if (parent)
+      parent->set_locked(false);
+    has_been_referenced = other.has_been_referenced;
+    parent = other.parent;
+    iter = std::move(other.iter);
+
+    other.parent = nullptr;
+  }
+  return *this;
+}
+
+simdjson_inline object_iterator::object_iterator(const object_iterator& other) noexcept
+  : has_been_referenced{other.has_been_referenced},
+    parent{nullptr},
+    iter{other.iter}
+{}
+
+simdjson_inline object_iterator& object_iterator::operator=(const object_iterator& other) noexcept {
+  if (this != &other)
+  {
+    if (parent)
+      parent->set_locked(false);
+    has_been_referenced = other.has_been_referenced;
+    parent = nullptr;
+    iter = other.iter;
+  }
+  return *this;
+}
+#endif
+
 simdjson_inline simdjson_result<field> object_iterator::operator*() noexcept {
 #if SIMDJSON_DEVELOPMENT_CHECKS
   // We must call * once per iteration.
@@ -142077,6 +181620,147 @@ simdjson_inline simdjson_result<westmere::ondemand::object_iterator> &simdjson_r

 #endif // SIMDJSON_GENERIC_ONDEMAND_OBJECT_ITERATOR_INL_H
 /* end file simdjson/generic/ondemand/object_iterator-inl.h for westmere */
+/* including simdjson/generic/ondemand/ranges-inl.h for westmere: #include "simdjson/generic/ondemand/ranges-inl.h" */
+/* begin file simdjson/generic/ondemand/ranges-inl.h for westmere */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/ranges.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array_iterator-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object_iterator-inl.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#if SIMDJSON_SUPPORTS_RANGES
+
+namespace simdjson {
+namespace westmere {
+namespace ondemand {
+
+//
+// array_range_iterator
+//
+
+simdjson_inline array_range_iterator::array_range_iterator(array_iterator iter) noexcept
+  : iter_{iter} {}
+
+simdjson_inline simdjson_result<value> array_range_iterator::operator*() const noexcept {
+  return *iter_;
+}
+
+simdjson_inline array_range_iterator& array_range_iterator::operator++() noexcept {
+  ++iter_;
+  return *this;
+}
+
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+simdjson_inline void array_range_iterator::operator++(int) noexcept {
+  ++*this;
+}
+SIMDJSON_POP_DISABLE_WARNINGS
+
+//
+// array_range
+//
+
+simdjson_inline array_range::array_range(array& arr) noexcept {
+  auto b = arr.begin();
+  if (b.error()) { error_ = b.error(); return; }
+  begin_ = b.value_unsafe();
+  end_ = arr.end().value_unsafe();
+}
+
+simdjson_inline array_range_iterator array_range::begin() noexcept {
+  return array_range_iterator(begin_);
+}
+
+simdjson_inline array_range_iterator array_range::end() noexcept {
+  return array_range_iterator(end_);
+}
+
+//
+// object_range_iterator
+//
+
+simdjson_inline object_range_iterator::object_range_iterator(object_iterator iter) noexcept
+  : iter_{iter} {}
+
+simdjson_inline simdjson_result<field> object_range_iterator::operator*() const noexcept {
+  return *iter_;
+}
+
+simdjson_inline object_range_iterator& object_range_iterator::operator++() noexcept {
+  ++iter_;
+  return *this;
+}
+
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+simdjson_inline void object_range_iterator::operator++(int) noexcept {
+  ++*this;
+}
+SIMDJSON_POP_DISABLE_WARNINGS
+
+
+//
+// object_range
+//
+
+simdjson_inline object_range::object_range(object& obj) noexcept {
+  auto b = obj.begin();
+  if (b.error()) { error_ = b.error(); return; }
+  begin_ = b.value_unsafe();
+  end_ = obj.end().value_unsafe();
+}
+
+simdjson_inline object_range_iterator object_range::begin() noexcept {
+  return object_range_iterator(begin_);
+}
+
+simdjson_inline object_range_iterator object_range::end() noexcept {
+  return object_range_iterator(end_);
+}
+
+//
+// Free functions
+//
+
+simdjson_inline array_range get_range(array& arr) noexcept {
+  return array_range(arr);
+}
+
+simdjson_inline object_range get_key_value_range(object& obj) noexcept {
+  return object_range(obj);
+}
+
+#if SIMDJSON_EXCEPTIONS
+simdjson_inline array_range get_range(simdjson_result<array> result) {
+  return array_range(result.value());
+}
+
+simdjson_inline object_range get_key_value_range(simdjson_result<object> result) {
+  return object_range(result.value());
+}
+#endif // SIMDJSON_EXCEPTIONS
+
+} // namespace ondemand
+} // namespace westmere
+} // namespace simdjson
+
+// Verify the range wrapper types satisfy the expected C++20 concepts.
+static_assert(std::input_iterator<simdjson::westmere::ondemand::array_range_iterator>);
+static_assert(std::input_iterator<simdjson::westmere::ondemand::object_range_iterator>);
+static_assert(std::ranges::input_range<simdjson::westmere::ondemand::array_range>);
+static_assert(std::ranges::input_range<simdjson::westmere::ondemand::object_range>);
+static_assert(std::ranges::view<simdjson::westmere::ondemand::array_range>);
+static_assert(std::ranges::view<simdjson::westmere::ondemand::object_range>);
+
+#endif // SIMDJSON_SUPPORTS_RANGES
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H
+/* end file simdjson/generic/ondemand/ranges-inl.h for westmere */
 /* including simdjson/generic/ondemand/parser-inl.h for westmere: #include "simdjson/generic/ondemand/parser-inl.h" */
 /* begin file simdjson/generic/ondemand/parser-inl.h for westmere */
 #ifndef SIMDJSON_GENERIC_ONDEMAND_PARSER_INL_H
@@ -142108,7 +181792,10 @@ simdjson_warn_unused simdjson_inline error_code parser::allocate(size_t new_capa

   // string_capacity copied from document::allocate
   _capacity = 0;
-  size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * new_capacity / 3 + SIMDJSON_PADDING, 64);
+  if(5 * (new_capacity / 3) + SIMDJSON_PADDING < SIMDJSON_PADDING) {
+    return CAPACITY; // overflow, only happen on legacy 32-bit systems with very large capacity
+  }
+  size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * (new_capacity / 3) + SIMDJSON_PADDING, 64);
   string_buf.reset(new (std::nothrow) uint8_t[string_capacity]);
 #if SIMDJSON_DEVELOPMENT_CHECKS
   start_positions.reset(new (std::nothrow) token_position[new_max_depth]);
@@ -142133,6 +181820,7 @@ simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate(p
   if (!json.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }

   json.remove_utf8_bom();
+  _document_len = json.length();

   // Allocate if needed
   if (capacity() < json.length() || !string_buf) {
@@ -142149,6 +181837,7 @@ simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate_a
   if (!json.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }

   json.remove_utf8_bom();
+  _document_len = json.length();

   // Allocate if needed
   if (capacity() < json.length() || !string_buf) {
@@ -142214,6 +181903,34 @@ simdjson_warn_unused simdjson_inline simdjson_result<json_iterator> parser::iter
   return json_iterator(reinterpret_cast<const uint8_t *>(json.data()), this);
 }

+inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size) noexcept {
+  if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
+  if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+    buf += 3;
+    len -= 3;
+  }
+  return document_stream(*this, buf, len, batch_size, false, stream_format::whitespace_delimited);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size) noexcept {
+  return iterate_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(padded_string_view s, size_t batch_size) noexcept {
+  if (!s.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
+  return iterate_many(s.data(), s.length(), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const padded_string &s, size_t batch_size) noexcept {
+  return iterate_many(padded_string_view(s), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const std::string &s, size_t batch_size) noexcept {
+  return iterate_many(padded_string_view(s), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size) noexcept {
+  return iterate_many(pad(s), batch_size);
+}
+
+#ifndef SIMDJSON_DISABLE_DEPRECATED_API
+SIMDJSON_PUSH_DISABLE_WARNINGS
+SIMDJSON_DISABLE_DEPRECATED_WARNING
 inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept {
   // Warning: no check is done on the buffer padding. We trust the user.
   if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
@@ -142221,8 +181938,11 @@ inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf,
     buf += 3;
     len -= 3;
   }
-  if(allow_comma_separated && batch_size < len) { batch_size = len; }
-  return document_stream(*this, buf, len, batch_size, allow_comma_separated);
+  // Map allow_comma_separated to stream_format::comma_delimited
+  if (allow_comma_separated) {
+    return document_stream(*this, buf, len, batch_size, false, stream_format::comma_delimited);
+  }
+  return document_stream(*this, buf, len, batch_size, false, stream_format::whitespace_delimited);
 }

 inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept {
@@ -142242,6 +181962,51 @@ inline simdjson_result<document_stream> parser::iterate_many(const std::string &
 inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size, bool allow_comma_separated) noexcept {
   return iterate_many(pad(s), batch_size, allow_comma_separated);
 }
+SIMDJSON_POP_DISABLE_WARNINGS
+#endif // SIMDJSON_DISABLE_DEPRECATED_API
+
+inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+  if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
+  if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+    buf += 3;
+    len -= 3;
+  }
+  if (format == stream_format::comma_delimited_array) {
+    // Strip leading JSON whitespace.
+    while (len > 0 && (buf[0] == ' ' || buf[0] == '\t' || buf[0] == '\n' || buf[0] == '\r')) {
+      buf++; len--;
+    }
+    // Expect the opening '['.
+    if (len == 0 || buf[0] != '[') { return TAPE_ERROR; }
+    buf++; len--;
+    // Strip trailing JSON whitespace.
+    while (len > 0 && (buf[len-1] == ' ' || buf[len-1] == '\t' || buf[len-1] == '\n' || buf[len-1] == '\r')) {
+      len--;
+    }
+    // Expect the closing ']'.
+    if (len == 0 || buf[len-1] != ']') { return TAPE_ERROR; }
+    len--;
+    // Fall through to comma_delimited over the array contents.
+    format = stream_format::comma_delimited;
+  }
+  return document_stream(*this, buf, len, batch_size, false, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+  return iterate_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(padded_string_view s, size_t batch_size, stream_format format) noexcept {
+  if (!s.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
+  return iterate_many(s.data(), s.length(), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const std::string &s, size_t batch_size, stream_format format) noexcept {
+  return iterate_many(padded_string_view(s), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size, stream_format format) noexcept {
+  return iterate_many(pad(s), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const padded_string &s, size_t batch_size, stream_format format) noexcept {
+  return iterate_many(padded_string_view(s), batch_size, format);
+}
 simdjson_pure simdjson_inline size_t parser::capacity() const noexcept {
   return _capacity;
 }
@@ -142649,6 +182414,27 @@ namespace simdjson {
 namespace westmere {
 namespace ondemand {

+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+simdjson_inline bool raw_json_string_is_quote_terminated(const uint8_t *json, uint32_t max_len) noexcept {
+  bool escaping{false};
+  for (uint32_t i = 1; i < max_len; i++) {
+    switch (json[i]) {
+      case '"':
+        if (!escaping) { return true; }
+        escaping = false;
+        break;
+      case '\\':
+        escaping = !escaping;
+        break;
+      default:
+        escaping = false;
+        break;
+    }
+  }
+  return false;
+}
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+
 simdjson_inline value_iterator::value_iterator(
   json_iterator *json_iter,
   depth_t depth,
@@ -143036,6 +182822,22 @@ simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iter
   return raw_json_string(key);
 }

+simdjson_warn_unused simdjson_inline error_code value_iterator::field_key_with_length(raw_json_string &key, std::size_t &len) noexcept {
+  assert_at_next();
+
+  const uint8_t *k = _json_iter->return_current_and_advance();
+  if (*(k++) != '"') { return report_error(TAPE_ERROR, "Object key is not a string"); }
+  // After return_current_and_advance(), the current token is the ':' that follows
+  // the key. The closing quote sits just before it (only JSON whitespace may
+  // intervene), so step back from the ':' to the closing quote to get the length.
+  // In minified JSON this is a single back-step.
+  const char *q = reinterpret_cast<const char *>(_json_iter->peek());
+  do { --q; } while (*q != '"');
+  key = raw_json_string(k);
+  len = static_cast<std::size_t>(q - reinterpret_cast<const char *>(k));
+  return SUCCESS;
+}
+
 simdjson_warn_unused simdjson_inline error_code value_iterator::field_value() noexcept {
   assert_at_next();

@@ -143153,7 +182955,7 @@ simdjson_warn_unused simdjson_inline error_code value_iterator::get_string(strin
   auto saved_string_buf_loc = _json_iter->string_buf_loc();
   auto err = get_string(allow_replacement).get(content);
   if (err) { return err; }
-  receiver = content;
+  internal::assign_utf8(receiver, content);
   // Restore the string buffer location, effectively discarding any temporary string storage
   _json_iter->string_buf_loc() = saved_string_buf_loc;
   return SUCCESS;
@@ -143164,6 +182966,15 @@ simdjson_warn_unused simdjson_inline simdjson_result<std::string_view> value_ite
 simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iterator::get_raw_json_string() noexcept {
   auto json = peek_scalar("string");
   if (*json != '"') { return incorrect_type_error("Not a string"); }
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  if (_json_iter->allow_incomplete_json()) {
+    const size_t remaining_input_length = _json_iter->remaining_input_length(json);
+    const uint32_t max_len = remaining_input_length < peek_start_length() ? uint32_t(remaining_input_length) : peek_start_length();
+    if (!raw_json_string_is_quote_terminated(json, max_len)) {
+      return STRING_ERROR;
+    }
+  }
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
   advance_scalar("string");
   return raw_json_string(json+1);
 }
@@ -143197,6 +183008,16 @@ simdjson_warn_unused simdjson_inline simdjson_result<double> value_iterator::get
   if(result.error() == SUCCESS) { advance_non_root_scalar("double"); }
   return result;
 }
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_float() noexcept {
+  auto result = numberparsing::parse_float(peek_non_root_scalar("float"));
+  if(result.error() == SUCCESS) { advance_non_root_scalar("float"); }
+  return result;
+}
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_float_in_string() noexcept {
+  auto result = numberparsing::parse_float_in_string(peek_non_root_scalar("float"));
+  if(result.error() == SUCCESS) { advance_non_root_scalar("float"); }
+  return result;
+}
 simdjson_warn_unused simdjson_inline simdjson_result<bool> value_iterator::get_bool() noexcept {
   auto result = parse_bool(peek_non_root_scalar("bool"));
   if(result.error() == SUCCESS) { advance_non_root_scalar("bool"); }
@@ -143299,7 +183120,7 @@ simdjson_warn_unused simdjson_inline error_code value_iterator::get_root_string(
   auto saved_string_buf_loc = _json_iter->string_buf_loc();
   auto err = get_root_string(check_trailing, allow_replacement).get(content);
   if (err) { return err; }
-  receiver = content;
+  internal::assign_utf8(receiver, content);
   // Restore the string buffer location, effectively discarding any temporary string storage
   _json_iter->string_buf_loc() = saved_string_buf_loc;
   return SUCCESS;
@@ -143311,6 +183132,15 @@ simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iter
   auto json = peek_scalar("string");
   if (*json != '"') { return incorrect_type_error("Not a string"); }
   if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  if (_json_iter->allow_incomplete_json()) {
+    const size_t remaining_input_length = _json_iter->remaining_input_length(json);
+    const uint32_t max_len = remaining_input_length < peek_root_length() ? uint32_t(remaining_input_length) : peek_root_length();
+    if (!raw_json_string_is_quote_terminated(json, max_len)) {
+      return STRING_ERROR;
+    }
+  }
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
   advance_scalar("string");
   return raw_json_string(json+1);
 }
@@ -143420,6 +183250,43 @@ simdjson_warn_unused simdjson_inline simdjson_result<double> value_iterator::get
   return result;
 }

+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_root_float(bool check_trailing) noexcept {
+  auto max_len = peek_root_length();
+  auto json = peek_root_scalar("float");
+  // We use the same buffer size as get_root_double: the number of significant
+  // digits that matter is smaller for binary32, but the JSON document may still
+  // spell out a long number that we must parse (and round) faithfully.
+  uint8_t tmpbuf[1074+8+1+1]; // +1 for null termination.
+  tmpbuf[1074+8+1] = '\0'; // make sure that buffer is always null terminated.
+  if (!_json_iter->copy_to_buffer(json, max_len, tmpbuf, 1074+8+1)) {
+    logger::log_error(*_json_iter, start_position(), depth(), "Root number more than 1082 characters");
+    return NUMBER_ERROR;
+  }
+  auto result = numberparsing::parse_float(tmpbuf);
+  if(result.error() == SUCCESS) {
+    if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+    advance_root_scalar("float");
+  }
+  return result;
+}
+
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_root_float_in_string(bool check_trailing) noexcept {
+  auto max_len = peek_root_length();
+  auto json = peek_root_scalar("float");
+  uint8_t tmpbuf[1074+8+1+1]; // +1 for null termination.
+  tmpbuf[1074+8+1] = '\0'; // make sure that buffer is always null terminated.
+  if (!_json_iter->copy_to_buffer(json, max_len, tmpbuf, 1074+8+1)) {
+    logger::log_error(*_json_iter, start_position(), depth(), "Root number more than 1082 characters");
+    return NUMBER_ERROR;
+  }
+  auto result = numberparsing::parse_float_in_string(tmpbuf);
+  if(result.error() == SUCCESS) {
+    if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+    advance_root_scalar("float");
+  }
+  return result;
+}
+
 simdjson_warn_unused simdjson_inline simdjson_result<bool> value_iterator::get_root_bool(bool check_trailing) noexcept {
   auto max_len = peek_root_length();
   auto json = peek_root_scalar("bool");
@@ -143648,6 +183515,21 @@ simdjson_inline void value_iterator::move_at_container_start() noexcept {
   _json_iter->token.set_position(_start_position + 1);
 }

+simdjson_inline void value_iterator::reenter_at(token_position position, depth_t depth) noexcept {
+  // Unlike reenter_child(), this does not require the live depth to be
+  // exactly one level shallower than depth, nor does it validate against
+  // the parser's per-depth container-start bookkeeping: neither holds in
+  // general for a caller-supplied snapshot (see object_position). What
+  // must still always hold, regardless of what was captured or how far
+  // the live iterator has since moved, is that position and depth are
+  // themselves sane values -- this is the same bound reenter_child()
+  // itself applies unconditionally.
+  SIMDJSON_ASSUME(position != nullptr);
+  SIMDJSON_ASSUME(depth >= 1 && depth < INT32_MAX);
+  _json_iter->_depth = depth;
+  _json_iter->token.set_position(position);
+}
+
 simdjson_inline simdjson_result<bool> value_iterator::reset_array() noexcept {
   if(error()) { return error(); }
   move_at_container_start();
@@ -145036,10 +184918,6 @@ simdjson_inline int count_ones(uint64_t input_num) {
   return __lsx_vpickve2gr_w(__lsx_vpcnt_d(__m128i(v2u64{input_num, 0})), 0);
 }

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2, uint64_t *result) {
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-}

 } // unnamed namespace
 } // namespace lsx
@@ -145575,7 +185453,7 @@ simdjson_inline escaping escaping::copy_and_find(const uint8_t *src, uint8_t *ds
 /* end file simdjson/lsx/begin.h */
 /* including simdjson/generic/ondemand/amalgamated.h for lsx: #include "simdjson/generic/ondemand/amalgamated.h" */
 /* begin file simdjson/generic/ondemand/amalgamated.h for lsx */
-#if defined(SIMDJSON_CONDITIONAL_INCLUDE) && !defined(SIMDJSON_GENERIC_BUILDER_DEPENDENCIES_H)
+#if defined(SIMDJSON_CONDITIONAL_INCLUDE) && !defined(SIMDJSON_GENERIC_ONDEMAND_DEPENDENCIES_H)
 #error simdjson/generic/ondemand/dependencies.h must be included before simdjson/generic/ondemand/amalgamated.h!
 #endif

@@ -145624,6 +185502,13 @@ class token_iterator;
 class value;
 class value_iterator;

+#if SIMDJSON_SUPPORTS_RANGES
+class array_range;
+class array_range_iterator;
+class object_range;
+class object_range_iterator;
+#endif // SIMDJSON_SUPPORTS_RANGES
+
 } // namespace ondemand
 } // namespace lsx
 } // namespace simdjson
@@ -145656,6 +185541,9 @@ template <> struct is_builtin_deserializable<lsx::ondemand::object> : std::true_
 template <> struct is_builtin_deserializable<lsx::ondemand::value> : std::true_type {};
 template <> struct is_builtin_deserializable<lsx::ondemand::raw_json_string> : std::true_type {};
 template <> struct is_builtin_deserializable<std::string_view> : std::true_type {};
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template <> struct is_builtin_deserializable<std::u8string_view> : std::true_type {};
+#endif // SIMDJSON_SUPPORTS_CHAR8_T

 template <typename T>
 concept is_builtin_deserializable_v = is_builtin_deserializable<T>::value;
@@ -145673,6 +185561,10 @@ concept nothrow_custom_deserializable = nothrow_tag_invocable<deserialize_tag, V
 template <typename T, typename ValT = lsx::ondemand::value>
 concept nothrow_deserializable = nothrow_custom_deserializable<T, ValT> || is_builtin_deserializable_v<T>;

+// True iff get<T>() on ValT cannot throw: no custom tag_invoke, or that tag_invoke is noexcept.
+template <typename T, typename ValT = lsx::ondemand::value>
+concept nothrow_gettable = !custom_deserializable<T, ValT> || nothrow_custom_deserializable<T, ValT>;
+
 /// Deserialize Tag
 inline constexpr struct deserialize_tag {
   using array_type = lsx::ondemand::array;
@@ -145887,6 +185779,17 @@ public:
    */
   simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> field_key() noexcept;

+  /**
+   * Get the current field's key together with its raw byte length.
+   *
+   * Like field_key(), but also returns the number of raw key bytes (the distance
+   * from the first key byte to the closing quote). The length is recovered from
+   * the structural index -- the next structural token is the ':' -- by stepping
+   * back over any whitespace to the closing quote, avoiding a forward SIMD scan
+   * for the closing quote. Leaves the iterator positioned exactly as field_key().
+   */
+  simdjson_warn_unused simdjson_inline error_code field_key_with_length(raw_json_string &key, std::size_t &len) noexcept;
+
   /**
    * Pass the : in the field and move to its value.
    */
@@ -146039,6 +185942,8 @@ public:
   simdjson_warn_unused simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_warn_unused simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_warn_unused simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<bool> get_bool() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<bool> is_null() noexcept;
   simdjson_warn_unused simdjson_inline bool is_negative() noexcept;
@@ -146057,6 +185962,8 @@ public:
   simdjson_warn_unused simdjson_inline simdjson_result<int64_t> get_root_int64_in_string(bool check_trailing) noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<double> get_root_double(bool check_trailing) noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<double> get_root_double_in_string(bool check_trailing) noexcept;
+  simdjson_warn_unused simdjson_inline simdjson_result<float> get_root_float(bool check_trailing) noexcept;
+  simdjson_warn_unused simdjson_inline simdjson_result<float> get_root_float_in_string(bool check_trailing) noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<bool> get_root_bool(bool check_trailing) noexcept;
   simdjson_warn_unused simdjson_inline bool is_root_negative() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<bool> is_root_integer(bool check_trailing) noexcept;
@@ -146192,6 +186099,15 @@ protected:

   /** @copydoc error_code json_iterator::position() const noexcept; */
   simdjson_inline token_position position() const noexcept;
+  /**
+   * Move the live iterator directly to the given position and depth, without
+   * validating against the parser's per-depth container-start bookkeeping
+   * (unlike json_iterator::reenter_child()). Used to restore a previously
+   * captured mid-container position (see object::revert_position()): that
+   * bookkeeping only tracks each container's own start, not every position
+   * a caller might later capture and revert to, so it does not apply here.
+   */
+  simdjson_inline void reenter_at(token_position position, depth_t depth) noexcept;
   /** @copydoc error_code json_iterator::end_position() const noexcept; */
   simdjson_inline token_position last_position() const noexcept;
   /** @copydoc error_code json_iterator::end_position() const noexcept; */
@@ -146260,9 +186176,12 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+   * Supported types: object, array, raw_json_string, string_view, u8string_view (C++20), uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool
    *
-   * You may use get_double(), get_bool(), get_uint64(), get_int64(),
+   * You may use get_double(), get_float(), get_bool(), get_uint64(), get_int64(),
+   * get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(),
+   * get_float32(), get_float64(),
    * get_object(), get_array(), get_raw_json_string(), or get_string() instead.
    * When SIMDJSON_SUPPORTS_CONCEPTS is set, custom types are also supported.
    *
@@ -146272,7 +186191,7 @@ public:
   template <typename T>
   simdjson_inline simdjson_result<T> get()
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+    noexcept(nothrow_gettable<T, value>)
 #else
     noexcept
 #endif
@@ -146287,7 +186206,8 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+   * Supported types: object, array, raw_json_string, string_view, u8string_view (C++20), uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool
    * If the macro SIMDJSON_SUPPORTS_CONCEPTS is set, then custom types are also supported.
    *
    * @param out This is set to a value of the given type, parsed from the JSON. If there is an error, this may not be initialized.
@@ -146297,7 +186217,7 @@ public:
   template <typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out)
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+    noexcept(nothrow_gettable<T, value>)
 #else
     noexcept
 #endif
@@ -146325,7 +186245,7 @@ public:
       "And you do not seem to have added support for it. Indeed, we have that "
       "simdjson::custom_deserializable<T> is false and the type T is not a default type "
       "such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-      "int64_t, double, or bool.");
+      "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
     static_cast<void>(out); // to get rid of unused errors
     return UNINITIALIZED;
   }
@@ -146334,7 +186254,8 @@ public:
     // immediately fail.
     static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
       "The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-      "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+      "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+      "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
       " get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
       " You may also add support for custom types, see our documentation.");
     static_cast<void>(out); // to get rid of unused errors
@@ -146412,6 +186333,50 @@ public:
    */
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;

+  /**
+   * Cast this JSON value to a 16-bit unsigned integer.
+   *
+   * Calls get_uint64() and checks that the result fits in a uint16_t.
+   *
+   * @returns A 16-bit unsigned integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint16_t.
+   */
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+
+  /**
+   * Cast this JSON value to a 16-bit signed integer.
+   *
+   * Calls get_int64() and checks that the result fits in an int16_t.
+   *
+   * @returns A 16-bit signed integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int16_t.
+   */
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+
+  /**
+   * Cast this JSON value to an 8-bit unsigned integer.
+   *
+   * Calls get_uint64() and checks that the result fits in a uint8_t.
+   *
+   * @returns An 8-bit unsigned integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint8_t.
+   */
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+
+  /**
+   * Cast this JSON value to an 8-bit signed integer.
+   *
+   * Calls get_int64() and checks that the result fits in an int8_t.
+   *
+   * @returns An 8-bit signed integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int8_t.
+   */
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
+
   /**
    * Cast this JSON value to a double.
    *
@@ -146428,6 +186393,53 @@ public:
    */
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;

+  /**
+   * Cast this JSON value to a float (binary32).
+   *
+   * The value is rounded directly to binary32: it is the float nearest to the
+   * JSON number. Note that this may differ from get_double() followed by a cast
+   * to float, which rounds twice.
+   *
+   * @returns A float.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+
+  /**
+   * Cast this JSON value (inside string) to a float (binary32).
+   *
+   * @returns A float.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  /**
+   * Cast this JSON value to a std::float32_t (C++23).
+   *
+   * Same as get_float(): the value is rounded directly to binary32.
+   *
+   * @returns A std::float32_t.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  /**
+   * Cast this JSON value to a std::float64_t (C++23).
+   *
+   * Same as get_double().
+   *
+   * @returns A std::float64_t.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   */
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
+
   /**
    * Cast this JSON value to a string.
    *
@@ -146455,6 +186467,26 @@ public:
    */
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;

+#if SIMDJSON_SUPPORTS_CHAR8_T
+  /**
+   * Cast this JSON value to a C++20 UTF-8 string.
+   *
+   * The string is guaranteed to be valid UTF-8.
+   *
+   * Equivalent to get<std::u8string_view>(). This is a zero-copy alias of
+   * get_string(): the very same bytes are returned, viewed as char8_t.
+   *
+   * Important: a value should be consumed once. Calling get_u8string() twice on the same
+   * value is an error.
+   *
+   * @param allow_replacement Whether to allow a replacement character for unmatched surrogate pairs.
+   * @returns An UTF-8 string. The string is stored in the parser and will be invalidated the next
+   *          time it parses a document or when it is destroyed.
+   * @returns INCORRECT_TYPE if the JSON value is not a string.
+   */
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
   /**
    * Attempts to fill the provided std::string reference with the parsed value of the current string.
    *
@@ -146542,7 +186574,7 @@ public:
   /**
    * Cast this JSON value to an unsigned integer.
    *
-   * @returns A signed 64-bit integer.
+   * @returns A unsigned 64-bit integer.
    * @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit unsigned integer.
    */
   simdjson_inline operator uint64_t() noexcept(false);
@@ -147007,9 +187039,24 @@ public:
   simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template <typename string_type>
   simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -147017,9 +187064,19 @@ public:
   simdjson_inline simdjson_result<bool> get_bool() noexcept;
   simdjson_inline simdjson_result<bool> is_null() noexcept;

-  template<typename T> simdjson_inline simdjson_result<T> get() noexcept;
+  template<typename T> simdjson_inline simdjson_result<T> get()
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lsx::ondemand::value>);
+#else
+    noexcept;
+#endif

-  template<typename T> simdjson_inline error_code get(T &out) noexcept;
+  template<typename T> simdjson_inline error_code get(T &out)
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lsx::ondemand::value>);
+#else
+    noexcept;
+#endif

 #if SIMDJSON_EXCEPTIONS
   template <class T>
@@ -147350,6 +187407,7 @@ protected:
   token_position _position{};

   friend class json_iterator;
+  friend class document_stream;
   friend class value_iterator;
   friend class object;
   template <typename... Args>
@@ -147441,6 +187499,9 @@ protected:
    * value of this attribute.
    */
   bool _streaming{false};
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  bool _allow_incomplete_json{false};
+#endif

 public:
   simdjson_inline json_iterator() noexcept = default;
@@ -147465,6 +187526,10 @@ public:
    * start_root_array() and start_root_object().
    */
   simdjson_inline bool streaming() const noexcept;
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  simdjson_inline bool allow_incomplete_json() const noexcept;
+  simdjson_inline size_t remaining_input_length(const uint8_t *json) const noexcept;
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON

   /**
    * Get the root value iterator
@@ -148344,33 +188409,87 @@ public:
    * @param batch_size The batch size to use. MUST be larger than the largest document. The sweet
    *                   spot is cache-related: small enough to fit in cache, yet big enough to
    *                   parse as many documents as possible in one tight loop.
-   *                   Defaults to 10MB, which has been a reasonable sweet spot in our tests.
-   * @param allow_comma_separated (defaults on false) This allows a mode where the documents are
-   *                   separated by commas instead of whitespace. It comes with a performance
-   *                   penalty because the entire document is indexed at once (and the document must be
-   *                   less than 4 GB), and there is no multithreading. In this mode, the batch_size parameter
-   *                   is effectively ignored, as it is set to at least the document size.
+   *                   Defaults to 1MB, which has been a reasonable sweet spot in our tests.
+   * @param allow_comma_separated @deprecated Use stream_format::comma_delimited instead.
+   *                   When true, maps internally to stream_format::comma_delimited.
+   *                   Defaults to false.
    * @return The stream, or an error. An empty input will yield 0 documents rather than an EMPTY error. Errors:
    *         - MEMALLOC if the parser does not have enough capacity and memory allocation fails
    *         - CAPACITY if the parser does not have enough capacity and batch_size > max_capacity.
    *         - other json errors if parsing fails. You should not rely on these errors to always the same for the
    *           same document: they may vary under runtime dispatch (so they may vary depending on your system and hardware).
    */
-  inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
-  inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
-  inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
-  inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size)
+  inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size)
     the string might be automatically padded with up to SIMDJSON_PADDING whitespace characters */
-  inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
-  inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
+  inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @private An rvalue input is destroyed at the end of the full-expression, while the
+   * returned document_stream only holds a pointer to it: iterating the stream would then
+   * read freed memory. These deleted overloads also catch a std::string_view argument,
+   * which would otherwise convert implicitly to a padded_string temporary. */
+  inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size = DEFAULT_BATCH_SIZE) = delete;// unsafe
+  /** @private @overload iterate_many(const std::string &&s, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size = DEFAULT_BATCH_SIZE) = delete;// unsafe
   /** @private We do not want to allow implicit conversion from C string to std::string. */
   simdjson_result<document_stream> iterate_many(const char *buf, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept = delete;

+#ifndef SIMDJSON_DISABLE_DEPRECATED_API
+  /**
+   * @deprecated Use iterate_many with stream_format::comma_delimited instead.
+   */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @private Deleted for the same reason as iterate_many(const std::string &&s, size_t batch_size). */
+  inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size, bool allow_comma_separated) = delete;// unsafe
+  /** @private @overload iterate_many(const std::string &&s, size_t batch_size, bool allow_comma_separated) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size, bool allow_comma_separated) = delete;// unsafe
+#endif // SIMDJSON_DISABLE_DEPRECATED_API
+
+  /**
+   * Parse a stream of JSON documents with explicit format specification.
+   *
+   * @param buf The concatenated JSON documents.
+   * @param len The length of the buffer.
+   * @param batch_size The batch size to use.
+   * @param format The stream format (whitespace_delimited, json_sequence, or comma_delimited).
+   * @return A stream of documents, or an error.
+   */
+  inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(padded_string_view s, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format)
+   *
+   * The string is padded in place if needed (see simdjson::pad), as with iterate_many(std::string &s, size_t batch_size).
+   */
+  inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size, stream_format format) noexcept;
+  /** @private Deleted for the same reason as iterate_many(const std::string &&s, size_t batch_size). */
+  inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+  /** @private @overload iterate_many(const std::string &&s, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+
   /** The capacity of this parser (the largest document it can process). */
   simdjson_pure simdjson_inline size_t capacity() const noexcept;
   /** The maximum capacity of this parser (the largest document it is allowed to process). */
@@ -148498,6 +188617,7 @@ private:
   size_t _capacity{0};
   size_t _max_capacity;
   size_t _max_depth{DEFAULT_MAX_DEPTH};
+  size_t _document_len{0};
   std::unique_ptr<uint8_t[]> string_buf{};

 #if SIMDJSON_DEVELOPMENT_CHECKS
@@ -148560,8 +188680,19 @@ public:
    * Begin array iteration.
    *
    * Part of the std::iterable interface.
+   *
+   * With SIMDJSON_DEVELOPMENT_CHECKS, the iterator locks this array while it is
+   * alive, so that reentrant access (at(), count_elements(), reset(), ...) is
+   * reported as OUT_OF_ORDER_ITERATION.
    */
-  simdjson_inline simdjson_result<array_iterator> begin() noexcept;
+  simdjson_inline simdjson_result<array_iterator> begin() & noexcept;
+  /**
+   * Begin iteration over a temporary array, e.g., `v.get_array().begin()`.
+   *
+   * The iterator does not depend on the array instance and may outlive it, so
+   * it does not lock it.
+   */
+  simdjson_inline simdjson_result<array_iterator> begin() && noexcept;
   /**
    * Sentinel representing the end of the array.
    *
@@ -148692,7 +188823,7 @@ public:
    */
   template <typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out)
-     noexcept(custom_deserializable<T, array> ? nothrow_custom_deserializable<T, array> : true) {
+     noexcept(nothrow_gettable<T, array>) {
     static_assert(custom_deserializable<T, array>);
     return deserialize(*this, out);
   }
@@ -148704,7 +188835,7 @@ public:
    */
   template <typename T>
   simdjson_inline simdjson_result<T> get()
-    noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+    noexcept(nothrow_gettable<T, array>)
   {
     static_assert(std::is_default_constructible<T>::value, "The specified type is not default constructible.");
     T out{};
@@ -148760,6 +188891,10 @@ protected:
    * iter.is_alive() == false indicates iteration is complete.
    */
   value_iterator iter{};
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  bool locked{false};
+  simdjson_inline void set_locked(bool _locked) noexcept;
+#endif

   friend class value;
   friend class document;
@@ -148781,7 +188916,8 @@ public:
   simdjson_inline simdjson_result(error_code error) noexcept; ///< @private
   simdjson_inline simdjson_result() noexcept = default;

-  simdjson_inline simdjson_result<lsx::ondemand::array_iterator> begin() noexcept;
+  simdjson_inline simdjson_result<lsx::ondemand::array_iterator> begin() & noexcept;
+  simdjson_inline simdjson_result<lsx::ondemand::array_iterator> begin() && noexcept;
   simdjson_inline simdjson_result<lsx::ondemand::array_iterator> end() noexcept;
   inline simdjson_result<size_t> count_elements() & noexcept;
   inline simdjson_result<bool> is_empty() & noexcept;
@@ -148801,7 +188937,7 @@ public:
   // TODO: move this code into object-inl.h

   template<typename T>
-  simdjson_inline simdjson_result<T> get() noexcept {
+  simdjson_inline simdjson_result<T> get() noexcept(nothrow_gettable<T, lsx::ondemand::array>) {
     if (error()) { return error(); }
     if constexpr (std::is_same_v<T, lsx::ondemand::array>) {
       return first;
@@ -148809,7 +188945,7 @@ public:
     return first.get<T>();
   }
   template<typename T>
-  simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept {
+  simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept(nothrow_gettable<T, lsx::ondemand::array>) {
     if (error()) { return error(); }
     if constexpr (std::is_same_v<T, lsx::ondemand::array>) {
       out = first;
@@ -148861,6 +188997,15 @@ public:
   /** Create a new, invalid array iterator. */
   simdjson_inline array_iterator() noexcept = default;

+#if SIMDJSON_DEVELOPMENT_CHECKS
+  simdjson_inline ~array_iterator() noexcept;
+
+  simdjson_inline array_iterator(array_iterator&&) noexcept;
+  simdjson_inline array_iterator& operator=(array_iterator&&) noexcept;
+  simdjson_inline array_iterator(const array_iterator&) noexcept;
+  simdjson_inline array_iterator& operator=(const array_iterator&) noexcept;
+#endif
+
   //
   // Iterator interface
   //
@@ -148903,6 +189048,9 @@ public:
 private:
 #if SIMDJSON_DEVELOPMENT_CHECKS
    bool has_been_referenced{false};
+   array* parent{nullptr};
+
+   simdjson_inline array_iterator(const value_iterator &_iter, array* _parent) noexcept;
 #endif
   value_iterator iter{};

@@ -149002,14 +189150,14 @@ public:
   /**
    * Cast this JSON value to an unsigned integer.
    *
-   * @returns A signed 64-bit integer.
+   * @returns A unsigned 64-bit integer.
    * @returns INCORRECT_TYPE If the JSON value is not a 64-bit unsigned integer.
    */
   simdjson_inline simdjson_result<uint64_t> get_uint64() noexcept;
   /**
    * Cast this JSON value (inside string) to an unsigned integer.
    *
-   * @returns A signed 64-bit integer.
+   * @returns A unsigned 64-bit integer.
    * @returns INCORRECT_TYPE If the JSON value is not a 64-bit unsigned integer.
    */
   simdjson_inline simdjson_result<uint64_t> get_uint64_in_string() noexcept;
@@ -149047,6 +189195,46 @@ public:
    * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int32_t.
    */
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  /**
+   * Cast this JSON value to a 16-bit unsigned integer.
+   *
+   * Calls get_uint64() and checks that the result fits in a uint16_t.
+   *
+   * @returns A 16-bit unsigned integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint16_t.
+   */
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  /**
+   * Cast this JSON value to a 16-bit signed integer.
+   *
+   * Calls get_int64() and checks that the result fits in an int16_t.
+   *
+   * @returns A 16-bit signed integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int16_t.
+   */
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  /**
+   * Cast this JSON value to an 8-bit unsigned integer.
+   *
+   * Calls get_uint64() and checks that the result fits in a uint8_t.
+   *
+   * @returns An 8-bit unsigned integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint8_t.
+   */
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  /**
+   * Cast this JSON value to an 8-bit signed integer.
+   *
+   * Calls get_int64() and checks that the result fits in an int8_t.
+   *
+   * @returns An 8-bit signed integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int8_t.
+   */
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   /**
    * Cast this JSON value to a double.
    *
@@ -149062,6 +189250,53 @@ public:
    * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
    */
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+
+  /**
+   * Cast this JSON value to a float (binary32).
+   *
+   * The value is rounded directly to binary32: it is the float nearest to the
+   * JSON number. Note that this may differ from get_double() followed by a cast
+   * to float, which rounds twice.
+   *
+   * @returns A float.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+
+  /**
+   * Cast this JSON value (inside string) to a float (binary32).
+   *
+   * @returns A float.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  /**
+   * Cast this JSON value to a std::float32_t (C++23).
+   *
+   * Same as get_float(): the value is rounded directly to binary32.
+   *
+   * @returns A std::float32_t.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  /**
+   * Cast this JSON value to a std::float64_t (C++23).
+   *
+   * Same as get_double().
+   *
+   * @returns A std::float64_t.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   */
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   /**
    * Cast this JSON value to a string.
    *
@@ -149075,6 +189310,24 @@ public:
    * @returns INCORRECT_TYPE if the JSON value is not a string.
    */
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  /**
+   * Cast this JSON value to a C++20 UTF-8 string.
+   *
+   * The string is guaranteed to be valid UTF-8.
+   *
+   * Equivalent to get<std::u8string_view>(). This is a zero-copy alias of
+   * get_string(): the very same bytes are returned, viewed as char8_t.
+   *
+   * Important: Calling get_u8string() twice on the same document is an error.
+   *
+   * @param allow_replacement Whether to allow a replacement character for unmatched surrogate pairs.
+   * @returns An UTF-8 string. The string is stored in the parser and will be invalidated the next
+   *          time it parses a document or when it is destroyed.
+   * @returns INCORRECT_TYPE if the JSON value is not a string.
+   */
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   /**
    * Attempts to fill the provided std::string reference with the parsed value of the current string.
    *
@@ -149145,9 +189398,12 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool
    *
-   * You may use get_double(), get_bool(), get_uint64(), get_int64(),
+   * You may use get_double(), get_float(), get_bool(), get_uint64(), get_int64(),
+   * get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(),
+   * get_float32(), get_float64(),
    * get_object(), get_array(), get_raw_json_string(), or get_string() instead.
    *
    * @returns A value of the given type, parsed from the JSON.
@@ -149156,7 +189412,7 @@ public:
   template <typename T>
   simdjson_inline simdjson_result<T> get() &
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    noexcept(nothrow_gettable<T, document>)
 #else
     noexcept
 #endif
@@ -149179,7 +189435,7 @@ public:
   template<typename T>
   simdjson_inline simdjson_result<T> get() &&
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    noexcept(nothrow_gettable<T, document>)
 #else
     noexcept
 #endif
@@ -149191,7 +189447,8 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool, value
+   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool, value
    *
    * Be mindful that the document instance must remain in scope while you are accessing object, array and value instances.
    *
@@ -149202,7 +189459,7 @@ public:
   template<typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out) &
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    noexcept(nothrow_gettable<T, document>)
 #else
     noexcept
 #endif
@@ -149215,7 +189472,7 @@ public:
         "And you do not seem to have added support for it. Indeed, we have that "
         "simdjson::custom_deserializable<T> is false and the type T is not a default type "
         "such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-        "int64_t, double, or bool.");
+        "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
       static_cast<void>(out); // to get rid of unused errors
       return UNINITIALIZED;
     }
@@ -149224,7 +189481,8 @@ public:
     // immediately fail.
     static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
       "The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-      "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+      "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+      "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
       " get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
       " You may also add support for custom types, see our documentation.");
     static_cast<void>(out); // to get rid of unused errors
@@ -149233,7 +189491,12 @@ public:
   }

   /** @overload template<typename T> error_code get(T &out) & noexcept */
-  template<typename T> simdjson_deprecated simdjson_inline error_code get(T &out) && noexcept;
+  template<typename T> simdjson_deprecated simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, document>);
+#else
+    noexcept;
+#endif

 #if SIMDJSON_EXCEPTIONS
   /**
@@ -149267,24 +189530,24 @@ public:
   /**
    * Cast this JSON value to an unsigned integer.
    *
-   * @returns A signed 64-bit integer.
+   * @returns A unsigned 64-bit integer.
    * @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit unsigned integer.
    */
-  simdjson_inline operator uint64_t() noexcept(false);
+  explicit simdjson_inline operator uint64_t() noexcept(false);
   /**
    * Cast this JSON value to a signed integer.
    *
    * @returns A signed 64-bit integer.
    * @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit integer.
    */
-  simdjson_inline operator int64_t() noexcept(false);
+  explicit simdjson_inline operator int64_t() noexcept(false);
   /**
    * Cast this JSON value to a double.
    *
    * @returns A double.
    * @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a valid floating-point number.
    */
-  simdjson_inline operator double() noexcept(false);
+  explicit simdjson_inline operator double() noexcept(false);
   /**
    * Cast this JSON value to a string.
    *
@@ -149294,7 +189557,7 @@ public:
    *          time it parses a document or when it is destroyed.
    * @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not a string.
    */
-  simdjson_inline operator std::string_view() noexcept(false) simdjson_lifetime_bound;
+  explicit simdjson_inline operator std::string_view() noexcept(false) simdjson_lifetime_bound;
   /**
    * Cast this JSON value to a raw_json_string.
    *
@@ -149303,14 +189566,14 @@ public:
    * @returns A pointer to the raw JSON for the given string.
    * @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not a string.
    */
-  simdjson_inline operator raw_json_string() noexcept(false) simdjson_lifetime_bound;
+  explicit simdjson_inline operator raw_json_string() noexcept(false) simdjson_lifetime_bound;
   /**
    * Cast this JSON value to a bool.
    *
    * @returns A bool value.
    * @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not true or false.
    */
-  simdjson_inline operator bool() noexcept(false);
+  explicit simdjson_inline operator bool() noexcept(false);
   /**
    * Cast this JSON value to a value when the document is an object or an array.
    *
@@ -149805,9 +190068,24 @@ public:
   simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template <typename string_type>
   simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -149819,7 +190097,7 @@ public:
   template <typename T>
   simdjson_inline simdjson_result<T> get() &
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    noexcept(nothrow_gettable<T, document_reference>)
 #else
     noexcept
 #endif
@@ -149832,7 +190110,8 @@ public:
   template<typename T>
   simdjson_inline simdjson_result<T> get() &&
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    // Forwards to document::get<T>(), so the document customization decides.
+    noexcept(nothrow_gettable<T, document>)
 #else
     noexcept
 #endif
@@ -149844,7 +190123,8 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool, value
+   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool, value
    *
    * Be mindful that the document instance must remain in scope while you are accessing object, array and value instances.
    *
@@ -149855,7 +190135,7 @@ public:
   template<typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out) &
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document_reference> : true)
+    noexcept(nothrow_gettable<T, document_reference>)
 #else
     noexcept
 #endif
@@ -149868,7 +190148,7 @@ public:
         "And you do not seem to have added support for it. Indeed, we have that "
         "simdjson::custom_deserializable<T> is false and the type T is not a default type "
         "such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-        "int64_t, double, or bool.");
+        "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
       static_cast<void>(out); // to get rid of unused errors
       return UNINITIALIZED;
     }
@@ -149877,7 +190157,8 @@ public:
     // immediately fail.
     static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
       "The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-      "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+      "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+      "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
       " get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
       " You may also add support for custom types, see our documentation.");
     static_cast<void>(out); // to get rid of unused errors
@@ -149886,7 +190167,12 @@ public:
   }

   /** @overload template<typename T> error_code get(T &out) & noexcept */
-  template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+  template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, document_reference>);
+#else
+    noexcept;
+#endif
   simdjson_inline simdjson_result<std::string_view> raw_json() noexcept;
 #if SIMDJSON_STATIC_REFLECTION
   template<constevalutil::fixed_string... FieldNames, typename T>
@@ -149899,12 +190185,12 @@ public:
   explicit simdjson_inline operator T() noexcept(false);
   simdjson_inline operator array() & noexcept(false);
   simdjson_inline operator object() & noexcept(false);
-  simdjson_inline operator uint64_t() noexcept(false);
-  simdjson_inline operator int64_t() noexcept(false);
-  simdjson_inline operator double() noexcept(false);
-  simdjson_inline operator std::string_view() noexcept(false);
-  simdjson_inline operator raw_json_string() noexcept(false);
-  simdjson_inline operator bool() noexcept(false);
+  explicit simdjson_inline operator uint64_t() noexcept(false);
+  explicit simdjson_inline operator int64_t() noexcept(false);
+  explicit simdjson_inline operator double() noexcept(false);
+  explicit simdjson_inline operator std::string_view() noexcept(false);
+  explicit simdjson_inline operator raw_json_string() noexcept(false);
+  explicit simdjson_inline operator bool() noexcept(false);
   simdjson_inline operator value() noexcept(false);
 #endif
   simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -149966,9 +190252,24 @@ public:
   simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template <typename string_type>
   simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -149977,11 +190278,31 @@ public:
   simdjson_inline simdjson_result<lsx::ondemand::value> get_value() noexcept;
   simdjson_inline simdjson_result<bool> is_null() noexcept;

-  template<typename T> simdjson_inline simdjson_result<T> get() & noexcept;
-  template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() && noexcept;
+  template<typename T> simdjson_inline simdjson_result<T> get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lsx::ondemand::document>);
+#else
+    noexcept;
+#endif
+  template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lsx::ondemand::document>);
+#else
+    noexcept;
+#endif

-  template<typename T> simdjson_inline error_code get(T &out) & noexcept;
-  template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+  template<typename T> simdjson_inline error_code get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lsx::ondemand::document>);
+#else
+    noexcept;
+#endif
+  template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lsx::ondemand::document>);
+#else
+    noexcept;
+#endif
 #if SIMDJSON_EXCEPTIONS

   using lsx::implementation_simdjson_result_base<lsx::ondemand::document>::operator*;
@@ -149990,12 +190311,12 @@ public:
   explicit simdjson_inline operator T() noexcept(false);
   simdjson_inline operator lsx::ondemand::array() & noexcept(false);
   simdjson_inline operator lsx::ondemand::object() & noexcept(false);
-  simdjson_inline operator uint64_t() noexcept(false);
-  simdjson_inline operator int64_t() noexcept(false);
-  simdjson_inline operator double() noexcept(false);
-  simdjson_inline operator std::string_view() noexcept(false);
-  simdjson_inline operator lsx::ondemand::raw_json_string() noexcept(false);
-  simdjson_inline operator bool() noexcept(false);
+  explicit simdjson_inline operator uint64_t() noexcept(false);
+  explicit simdjson_inline operator int64_t() noexcept(false);
+  explicit simdjson_inline operator double() noexcept(false);
+  explicit simdjson_inline operator std::string_view() noexcept(false);
+  explicit simdjson_inline operator lsx::ondemand::raw_json_string() noexcept(false);
+  explicit simdjson_inline operator bool() noexcept(false);
   simdjson_inline operator lsx::ondemand::value() noexcept(false);
 #endif
   simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -150061,9 +190382,24 @@ public:
   simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template <typename string_type>
   simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -150072,22 +190408,42 @@ public:
   simdjson_inline simdjson_result<lsx::ondemand::value> get_value() noexcept;
   simdjson_inline simdjson_result<bool> is_null() noexcept;

-  template<typename T> simdjson_inline simdjson_result<T> get() & noexcept;
-  template<typename T> simdjson_inline simdjson_result<T> get() && noexcept;
+  template<typename T> simdjson_inline simdjson_result<T> get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lsx::ondemand::document_reference>);
+#else
+    noexcept;
+#endif
+  template<typename T> simdjson_inline simdjson_result<T> get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lsx::ondemand::document_reference>);
+#else
+    noexcept;
+#endif

-  template<typename T> simdjson_inline error_code get(T &out) & noexcept;
-  template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+  template<typename T> simdjson_inline error_code get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lsx::ondemand::document_reference>);
+#else
+    noexcept;
+#endif
+  template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lsx::ondemand::document_reference>);
+#else
+    noexcept;
+#endif
 #if SIMDJSON_EXCEPTIONS
   template <class T>
   explicit simdjson_inline operator T() noexcept(false);
   simdjson_inline operator lsx::ondemand::array() & noexcept(false);
   simdjson_inline operator lsx::ondemand::object() & noexcept(false);
-  simdjson_inline operator uint64_t() noexcept(false);
-  simdjson_inline operator int64_t() noexcept(false);
-  simdjson_inline operator double() noexcept(false);
-  simdjson_inline operator std::string_view() noexcept(false);
-  simdjson_inline operator lsx::ondemand::raw_json_string() noexcept(false);
-  simdjson_inline operator bool() noexcept(false);
+  explicit simdjson_inline operator uint64_t() noexcept(false);
+  explicit simdjson_inline operator int64_t() noexcept(false);
+  explicit simdjson_inline operator double() noexcept(false);
+  explicit simdjson_inline operator std::string_view() noexcept(false);
+  explicit simdjson_inline operator lsx::ondemand::raw_json_string() noexcept(false);
+  explicit simdjson_inline operator bool() noexcept(false);
   simdjson_inline operator lsx::ondemand::value() noexcept(false);
 #endif
   simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -150255,10 +190611,7 @@ public:
    *   }
    *   size_t truncated = stream.truncated_bytes();
    *
-   * IMPORTANT: this value is only meaningful under the conditions below. It is
-   * computed from stage-1 bookkeeping, and outside these conditions it is not
-   * merely imprecise, it is arbitrary -- it can exceed size_in_bytes() or wrap
-   * around to a huge value. Check it only when both of the following hold:
+   * IMPORTANT: this value is only meaningful under the conditions below.
    *
    *   - you iterated all the way to the end of the stream;
    *   - no document reported an error. Iteration stops at the first failed
@@ -150267,6 +190620,9 @@ public:
    * If you need to know about a truncated tail outside those conditions, track
    * it yourself from the last successful document (see iterator::current_index()
    * and iterator::source()).
+   *
+   * An empty input (zero bytes) or an input made only of white space contains
+   * no document: truncated_bytes() returns zero.
    */
   inline size_t truncated_bytes() const noexcept;

@@ -150326,7 +190682,10 @@ public:
      *
      * The returned string_view instance is simply a map to the (unparsed)
      * source string: it may thus include white-space characters and all manner
-     * of padding.
+     * of padding. It spans the whole current document, whether or not you
+     * have already accessed (part of) the document. Thus
+     * current_index() + source().size() is the offset just past the end of the
+     * current document, which is useful when reading a stream in chunks.
      *
      * This function (source()) is experimental and the usage
      * may change in future versions of simdjson: we find the API somewhat
@@ -150380,13 +190739,16 @@ private:
    * @param buf is the raw byte buffer we need to process
    * @param len is the length of the raw byte buffer in bytes
    * @param batch_size is the size of the windows (must be strictly greater or equal to the largest JSON document)
+   * @param allow_comma_separated whether to allow comma-separated documents
+   * @param format the stream format
    */
   simdjson_inline document_stream(
     ondemand::parser &parser,
     const uint8_t *buf,
     size_t len,
     size_t batch_size,
-    bool allow_comma_separated
+    bool allow_comma_separated,
+    stream_format format = stream_format::whitespace_delimited
   ) noexcept;

   /**
@@ -150420,8 +190782,23 @@ private:
    */
   inline void next() noexcept;

-  /** Move the json_iterator of the document to the location of the next document in the stream. */
+  /**
+   * Move the json_iterator of the document to the location of the next document
+   * in the stream.
+   *
+   * For formats with a document delimiter (`newline_delimited`, `json_sequence`),
+   * when the iterator is still inside the current document (`depth() > 0`), this
+   * may jump to the next delimiter instead of walking remaining structurals. That
+   * jump does not structure-validate the unread remainder.
+   */
   inline void next_document() noexcept;
+  /** Byte that ends a document under `format`, or 0 if there is none. */
+  simdjson_inline uint8_t document_delimiter() const noexcept;
+  /**
+   * Position the iterator at the first structural at or past the next
+   * `delimiter` in the current batch. Returns false if none is found.
+   */
+  simdjson_inline bool skip_to_delimiter(uint8_t delimiter) noexcept;

   /** Get the next document index. */
   inline size_t next_batch_start() const noexcept;
@@ -150435,6 +190812,7 @@ private:
   size_t len;
   size_t batch_size;
   bool allow_comma_separated;
+  stream_format format;
   /**
    * We are going to use just one document instance. The document owns
    * the json_iterator. It implies that we only ever pass a reference
@@ -150461,7 +190839,7 @@ private:
   /** The error returned from the stage 1 thread. */
   error_code stage1_thread_error{UNINITIALIZED};
   /** The thread used to run stage 1 against the next batch in the background. */
-  std::unique_ptr<stage1_worker> worker{new(std::nothrow) stage1_worker()};
+  std::unique_ptr<stage1_worker> worker{};
   /**
    * The parser used to run stage 1 in the background. Will be swapped
    * with the regular parser when finished.
@@ -150536,6 +190914,16 @@ public:
    * call it again nor can you call key().
    */
   simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  /**
+   * Get the key as a C++20 u8string_view. This is a zero-copy alias of
+   * unescaped_key(): the very same bytes are returned, viewed as char8_t.
+   *
+   * This consumes the key: once you have called unescaped_u8key(), you cannot
+   * call it again nor can you call key().
+   */
+  simdjson_inline simdjson_warn_unused simdjson_result<std::u8string_view> unescaped_u8key(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   /**
    * Get the key as a string_view (for higher speed, consider raw_key).
    * We deliberately use a more cumbersome name (unescaped_key) to force users
@@ -150568,6 +190956,16 @@ public:
    * you can safely call it repeatedly. Use unescaped_key() to get the unescaped key.
    */
   simdjson_inline std::string_view escaped_key() const noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  /**
+   * Get the key as a C++20 u8string_view. This is a zero-copy alias of
+   * escaped_key(): the very same bytes are returned, viewed as char8_t.
+   * The string is unprocessed, so it may contain escape characters
+   * (e.g., \uXXXX or \n). It does not count as a consumption of the content:
+   * you can safely call it repeatedly.
+   */
+  simdjson_inline std::u8string_view escaped_u8key() const noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   /**
    * Get the field value.
    */
@@ -150599,11 +190997,17 @@ public:
   simdjson_inline simdjson_result() noexcept = default;

   simdjson_inline simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> unescaped_u8key(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template<typename string_type>
   simdjson_inline error_code unescaped_key(string_type &receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<lsx::ondemand::raw_json_string> key() noexcept;
   simdjson_inline simdjson_result<std::string_view> key_raw_json_token() noexcept;
   simdjson_inline simdjson_result<std::string_view> escaped_key() noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> escaped_u8key() noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   simdjson_inline simdjson_result<lsx::ondemand::value> value() noexcept;
 };

@@ -150611,6 +191015,1398 @@ public:

 #endif // SIMDJSON_GENERIC_ONDEMAND_FIELD_H
 /* end file simdjson/generic/ondemand/field.h for lsx */
+/* including simdjson/generic/ondemand/key_selector.h for lsx: #include "simdjson/generic/ondemand/key_selector.h" */
+/* begin file simdjson/generic/ondemand/key_selector.h for lsx */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+#define SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #include "simdjson/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/common_defs.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/constevalutil.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/raw_json_string.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#include <array>
+#include <string>      // std::string (key_selector::describe)
+#include <string_view>
+#include <cstddef>
+#include <cstdint>
+#include <cstring>     // std::memcpy (portable unaligned window load)
+#include <utility>     // std::index_sequence (window candidate dispatch)
+#include <type_traits> // std::integral_constant (window candidate dispatch)
+
+// AArch64 only: 32-bit ARM (ARMv7) defines __ARM_NEON too, but lacks the
+// A64-only horizontal reductions (vmaxvq_u32) used below. See issue #2885.
+#if defined(__aarch64__)
+  #include <arm_neon.h>
+  #define SIMDJSON_KEY_SELECTOR_HAS_NEON 1
+#else
+  #define SIMDJSON_KEY_SELECTOR_HAS_NEON 0
+#endif
+#if defined(__SSE2__)
+  #include <emmintrin.h>
+  #define SIMDJSON_KEY_SELECTOR_HAS_SSE2 1
+#else
+  #define SIMDJSON_KEY_SELECTOR_HAS_SSE2 0
+#endif
+#if defined(__loongarch_sx)
+  #include <lsxintrin.h>
+  #define SIMDJSON_KEY_SELECTOR_HAS_LSX 1
+#else
+  #define SIMDJSON_KEY_SELECTOR_HAS_LSX 0
+#endif
+
+#if SIMDJSON_SUPPORTS_CONCEPTS
+
+namespace simdjson {
+namespace lsx {
+namespace ondemand {
+
+namespace key_selector_detail {
+
+// Since not constexpr, triggers compile-time error.
+inline void compile_time_error(const char* message) noexcept { (void)message; }
+
+// ============================================================================
+// Compile-time perfect-hash generator.
+// It scales to ~100 keys at compile time by determining association values one (position, character)
+// symbol at a time (gperf-style) instead of an exhaustive offset search, and
+// falls back to a Hash-and-Displace construction for large/awkward key sets.
+//
+// Only flat tables survive to runtime; the lookup is a few additions plus a
+// single SIMD key comparison (see match_raw below).
+// ============================================================================
+
+// Maximum number of character positions the gperf hash may combine.
+static constexpr std::size_t MAX_POSITIONS = 16;
+// Sentinel "position" meaning "the last character of the key".
+static constexpr std::size_t LAST_CHAR = std::size_t(-1);
+// Runtime-encoded sentinels (stored in uint8 tables).
+static constexpr std::uint8_t POS_LAST_CHAR = 0xFF; // positions_[i] == last char
+static constexpr std::uint8_t HD_MODE       = 0xFF; // num_positions == H&D mode
+// Flags stored in positions[2] in H&D mode to select the key-hash variant.
+static constexpr std::size_t HD_HASH_2BYTE_FLAG = 2;
+static constexpr std::size_t HD_HASH_4BYTE_FLAG = 4;
+
+constexpr std::size_t next_power_of_2(std::size_t n) noexcept {
+    if (n == 0) { return 1; }
+    std::size_t p = 1;
+    while (p < n) { p <<= 1; }
+    return p;
+}
+
+// Character at a given position (LAST_CHAR means last character), or 256 if out
+// of bounds.
+constexpr std::size_t char_at(std::string_view key, std::size_t pos) noexcept {
+    if (pos == LAST_CHAR) {
+        if (key.empty()) { return 256; }
+        return static_cast<unsigned char>(key[key.size() - 1]);
+    }
+    if (pos >= key.size()) { return 256; }
+    return static_cast<unsigned char>(key[pos]);
+}
+
+// Count key pairs that a set of positions fails to distinguish. Keys whose
+// lengths differ modulo the table size are separated by the length term in the
+// hash, so they need no position coverage.
+template <std::size_t N>
+consteval std::size_t count_undistinguished_pairs(
+    const std::array<std::string_view, N>& keys,
+    const std::size_t* positions,
+    std::size_t num_positions,
+    std::size_t modulus) {
+    std::size_t count = 0;
+    for (std::size_t i = 0; i < N; ++i) {
+        for (std::size_t j = i + 1; j < N; ++j) {
+            if (keys[i].size() % modulus != keys[j].size() % modulus) { continue; }
+            bool distinguished = false;
+            for (std::size_t p = 0; p < num_positions; ++p) {
+                if (char_at(keys[i], positions[p]) != char_at(keys[j], positions[p])) {
+                    distinguished = true;
+                    break;
+                }
+            }
+            if (!distinguished) { ++count; }
+        }
+    }
+    return count;
+}
+
+template <std::size_t N>
+consteval bool positions_distinguish(
+    const std::array<std::string_view, N>& keys,
+    const std::size_t* positions,
+    std::size_t num_positions,
+    std::size_t modulus) {
+    return count_undistinguished_pairs<N>(keys, positions, num_positions, modulus) == 0;
+}
+
+// Number of distinct (length % modulus, char_at(key, pos)) pairs at a position.
+template <std::size_t N>
+consteval std::size_t discriminating_power(
+    const std::array<std::string_view, N>& keys,
+    std::size_t pos,
+    std::size_t modulus) {
+    struct pair { std::size_t len_mod; std::size_t ch; };
+    std::array<pair, N> pairs{};
+    for (std::size_t i = 0; i < N; ++i) {
+        pairs[i] = {keys[i].size() % modulus, char_at(keys[i], pos)};
+    }
+    std::size_t count = 0;
+    for (std::size_t i = 0; i < N; ++i) {
+        bool dup = false;
+        for (std::size_t j = 0; j < i; ++j) {
+            if (pairs[i].len_mod == pairs[j].len_mod && pairs[i].ch == pairs[j].ch) {
+                dup = true;
+                break;
+            }
+        }
+        if (!dup) { ++count; }
+    }
+    return count;
+}
+
+template <std::size_t N>
+consteval std::size_t max_key_length(const std::array<std::string_view, N>& keys) {
+    std::size_t m = 0;
+    for (std::size_t i = 0; i < N; ++i) {
+        if (keys[i].size() > m) { m = keys[i].size(); }
+    }
+    return m;
+}
+
+// Bounded backtracking DFS for a minimal set of distinguishing positions.
+template <std::size_t N>
+consteval bool backtracking_search(
+    const std::array<std::string_view, N>& keys,
+    const std::size_t* candidates,
+    std::size_t num_candidates,
+    std::size_t* positions,
+    std::size_t& num_positions_out,
+    std::size_t& budget,
+    std::size_t modulus) {
+    constexpr std::size_t MAX_DEPTH = 8;
+    std::size_t breadth = num_candidates < 20 ? num_candidates : 20;
+
+    struct frame { std::size_t depth; std::size_t next_ci; std::size_t parent_count; };
+    std::array<frame, MAX_DEPTH + 1> stack{};
+    std::size_t sp = 0;
+
+    std::size_t initial_count = count_undistinguished_pairs<N>(keys, positions, 0, modulus);
+    if (budget > 0) { --budget; }
+    if (initial_count == 0) { num_positions_out = 0; return true; }
+
+    stack[0] = {0, 0, initial_count};
+
+    while (budget > 0) {
+        if (sp > MAX_DEPTH) {
+            if (sp == 0) { break; }
+            --sp;
+            ++stack[sp].next_ci;
+            continue;
+        }
+        auto& f = stack[sp];
+        if (f.next_ci >= breadth) {
+            if (sp == 0) { break; }
+            --sp;
+            ++stack[sp].next_ci;
+            continue;
+        }
+        positions[sp] = candidates[f.next_ci];
+        --budget;
+        std::size_t new_count = count_undistinguished_pairs<N>(keys, positions, sp + 1, modulus);
+        if (new_count == 0) { num_positions_out = sp + 1; return true; }
+        if (new_count < f.parent_count && sp + 1 < MAX_DEPTH) {
+            stack[sp + 1] = {sp + 1, f.next_ci + 1, new_count};
+            ++sp;
+        } else {
+            ++f.next_ci;
+        }
+    }
+    return false;
+}
+
+// Phase 1: select character positions that distinguish all colliding pairs.
+template <std::size_t N>
+consteval std::size_t select_positions(
+    const std::array<std::string_view, N>& keys,
+    std::array<std::size_t, MAX_POSITIONS>& positions,
+    std::size_t modulus) {
+    if (positions_distinguish<N>(keys, positions.data(), 0, modulus)) { return 0; }
+
+    std::size_t max_len = max_key_length(keys);
+    constexpr std::size_t MAX_CANDIDATES = 256;
+    std::array<std::size_t, MAX_CANDIDATES> candidates{};
+    std::array<std::size_t, MAX_CANDIDATES> powers{};
+    std::size_t num_candidates = 0;
+    for (std::size_t p = 0; p < max_len && num_candidates < MAX_CANDIDATES - 1; ++p) {
+        candidates[num_candidates] = p;
+        powers[num_candidates] = discriminating_power(keys, p, modulus);
+        ++num_candidates;
+    }
+    if (num_candidates < MAX_CANDIDATES) {
+        candidates[num_candidates] = LAST_CHAR;
+        powers[num_candidates] = discriminating_power(keys, LAST_CHAR, modulus);
+        ++num_candidates;
+    }
+    for (std::size_t i = 0; i < num_candidates; ++i) {
+        for (std::size_t j = i + 1; j < num_candidates; ++j) {
+            if (powers[j] > powers[i]) {
+                auto tc = candidates[i]; candidates[i] = candidates[j]; candidates[j] = tc;
+                auto tp = powers[i]; powers[i] = powers[j]; powers[j] = tp;
+            }
+        }
+    }
+
+    positions[0] = candidates[0];
+    if (positions_distinguish<N>(keys, positions.data(), 1, modulus)) { return 1; }
+
+    positions[0] = 0;
+    positions[1] = LAST_CHAR;
+    if (positions_distinguish<N>(keys, positions.data(), 2, modulus)) { return 2; }
+
+    {
+        std::size_t budget = 5000;
+        std::size_t num_found = 0;
+        if (backtracking_search<N>(keys, candidates.data(), num_candidates,
+                                   positions.data(), num_found, budget, modulus)) {
+            return num_found;
+        }
+    }
+
+    std::size_t num_pos = 0;
+    for (std::size_t ci = 0; ci < num_candidates && num_pos < MAX_POSITIONS; ++ci) {
+        bool already = false;
+        for (std::size_t p = 0; p < num_pos; ++p) {
+            if (positions[p] == candidates[ci]) { already = true; break; }
+        }
+        if (already) { continue; }
+        positions[num_pos] = candidates[ci];
+        ++num_pos;
+        if (positions_distinguish<N>(keys, positions.data(), num_pos, modulus)) { return num_pos; }
+    }
+
+    compile_time_error("Failed to find distinguishing positions for perfect hash");
+    return 0;
+}
+
+// Result of PHF computation. A max-sized slot_to_key array lets the same struct
+// type carry any chosen table size.
+template <std::size_t N>
+struct phf_result {
+    // Allow up to 8x the minimum table size. Sparser tables solve faster.
+    static constexpr std::size_t MAX_TABLE_SIZE = next_power_of_2(N) * 8;
+    std::size_t table_size{};
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso_values{};
+    std::size_t num_positions{};
+    std::array<std::size_t, MAX_POSITIONS> positions{};
+    std::array<std::size_t, MAX_TABLE_SIZE> slot_to_key{};
+};
+
+// Partition-based asso_values search (gperf-style). Determines asso_values one
+// (position, character) symbol at a time; never revisits a value. Equivalence
+// classes (keys sharing the same undetermined symbols) keep the search cheap.
+template <std::size_t N, std::size_t M>
+consteval bool try_generate_gperf(
+    const std::array<std::string_view, N>& keys,
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS>& asso_values,
+    std::size_t& num_positions,
+    std::array<std::size_t, MAX_POSITIONS>& positions,
+    std::array<std::size_t, M>& slot_to_key) {
+    num_positions = select_positions<N>(keys, positions, M);
+
+    for (std::size_t p = 0; p < MAX_POSITIONS; ++p) {
+        for (std::size_t c = 0; c < 256; ++c) { asso_values[p][c] = 0; }
+    }
+
+    if (num_positions == 0) {
+        for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+        for (std::size_t i = 0; i < N; ++i) {
+            std::size_t slot = keys[i].size() % M;
+            if (slot_to_key[slot] != N) { return false; }
+            slot_to_key[slot] = i;
+        }
+        return true;
+    }
+
+    std::array<std::array<std::size_t, MAX_POSITIONS>, N> kchars{};
+    for (std::size_t k = 0; k < N; ++k) {
+        for (std::size_t p = 0; p < num_positions; ++p) {
+            kchars[k][p] = char_at(keys[k], positions[p]);
+        }
+    }
+
+    struct sym_t { std::size_t pos; std::size_t ch; std::size_t freq; };
+    constexpr std::size_t MAX_SYMS = MAX_POSITIONS * 256;
+    std::array<sym_t, MAX_SYMS> syms{};
+    std::size_t nsyms = 0;
+    for (std::size_t p = 0; p < num_positions; ++p) {
+        std::array<std::size_t, 256> freq{};
+        for (std::size_t k = 0; k < N; ++k) {
+            std::size_t c = kchars[k][p];
+            if (c < 256) { freq[c]++; }
+        }
+        for (std::size_t c = 0; c < 256; ++c) {
+            if (freq[c] > 0) { syms[nsyms++] = {p, c, freq[c]}; }
+        }
+    }
+    for (std::size_t i = 0; i < nsyms; ++i) {
+        for (std::size_t j = i + 1; j < nsyms; ++j) {
+            if (syms[j].freq > syms[i].freq) {
+                auto tmp = syms[i]; syms[i] = syms[j]; syms[j] = tmp;
+            }
+        }
+    }
+
+    std::array<std::size_t, N> phash{};
+    for (std::size_t k = 0; k < N; ++k) { phash[k] = keys[k].size(); }
+
+    std::array<std::array<uint64_t, 256>, MAX_POSITIONS> salt{};
+    {
+        uint64_t s = 0x9e3779b97f4a7c15ULL;
+        for (std::size_t p = 0; p < num_positions; ++p) {
+            for (std::size_t c = 0; c < 256; ++c) {
+                s = s * 6364136223846793005ULL + 1442695040888963407ULL;
+                salt[p][c] = s;
+            }
+        }
+    }
+    std::array<uint64_t, N> sig{};
+    for (std::size_t k = 0; k < N; ++k) {
+        uint64_t s = 0;
+        for (std::size_t p = 0; p < num_positions; ++p) {
+            std::size_t c = kchars[k][p];
+            if (c < 256) { s ^= salt[p][c]; }
+        }
+        sig[k] = s;
+    }
+    std::array<std::size_t, N> order{};
+    for (std::size_t k = 0; k < N; ++k) { order[k] = k; }
+
+    std::array<std::size_t, M> slot_gen{};
+    std::size_t gen = 0;
+
+    std::size_t search_limit = next_power_of_2(M);
+    if (search_limit < 32) { search_limit = 32; }
+
+    for (std::size_t si = 0; si < nsyms; ++si) {
+        std::size_t sp = syms[si].pos;
+        std::size_t sc = syms[si].ch;
+
+        uint64_t sp_salt = salt[sp][sc];
+        for (std::size_t k = 0; k < N; ++k) {
+            if (kchars[k][sp] == sc) { sig[k] ^= sp_salt; }
+        }
+
+        for (std::size_t i = 1; i < N; ++i) {
+            std::size_t x = order[i];
+            uint64_t xs = sig[x];
+            std::size_t j = i;
+            while (j > 0 && sig[order[j - 1]] > xs) {
+                order[j] = order[j - 1];
+                --j;
+            }
+            order[j] = x;
+        }
+
+        bool found = false;
+        for (std::size_t v = 0; v < search_limit && !found; ++v) {
+            bool collision = false;
+            std::size_t ci = 0;
+            while (ci < N && !collision) {
+                uint64_t class_sig = sig[order[ci]];
+                std::size_t cj = ci;
+                while (cj < N && sig[order[cj]] == class_sig) { ++cj; }
+                if (cj - ci > 1) {
+                    ++gen;
+                    for (std::size_t x = ci; x < cj; ++x) {
+                        std::size_t k = order[x];
+                        std::size_t h = phash[k];
+                        if (kchars[k][sp] == sc) { h += v; }
+                        h %= M;
+                        if (slot_gen[h] == gen) { collision = true; break; }
+                        slot_gen[h] = gen;
+                    }
+                }
+                ci = cj;
+            }
+            if (!collision) {
+                asso_values[sp][sc] = v;
+                for (std::size_t k = 0; k < N; ++k) {
+                    if (kchars[k][sp] == sc) { phash[k] += v; }
+                }
+                found = true;
+            }
+        }
+        if (!found) { return false; }
+    }
+
+    for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+    for (std::size_t i = 0; i < N; ++i) {
+        std::size_t slot = phash[i] % M;
+        if (slot_to_key[slot] != N) { return false; }
+        slot_to_key[slot] = i;
+    }
+    std::size_t filled = 0;
+    for (std::size_t i = 0; i < M; ++i) {
+        if (slot_to_key[i] != N) { ++filled; }
+    }
+    return filled == N;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_compute_phf(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+    static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso{};
+    std::size_t npos{};
+    std::array<std::size_t, MAX_POSITIONS> pos{};
+    std::array<std::size_t, M> s2k{};
+    if (try_generate_gperf<N, M>(keys, asso, npos, pos, s2k)) {
+        result.table_size = M;
+        result.asso_values = asso;
+        result.num_positions = npos;
+        result.positions = pos;
+        for (std::size_t i = 0; i < M; ++i) { result.slot_to_key[i] = s2k[i]; }
+        for (std::size_t i = M; i < phf_result<N>::MAX_TABLE_SIZE; ++i) { result.slot_to_key[i] = N; }
+        return true;
+    }
+    return false;
+}
+
+template <std::size_t N, std::size_t M, std::size_t MaxM>
+consteval bool try_gperf_po2(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+    if (try_compute_phf<N, M>(keys, result)) { return true; }
+    constexpr std::size_t NextM = M * 2;
+    if constexpr (NextM <= MaxM) { return try_gperf_po2<N, NextM, MaxM>(keys, result); }
+    return false;
+}
+
+// --- Hash-and-Displace fallback --------------------------------------------
+
+constexpr std::size_t hd_bucket_hash(std::string_view key) noexcept {
+    std::size_t c0 = key.empty() ? 0 : static_cast<unsigned char>(key[0]);
+    std::size_t c1 = key.empty() ? 0 : static_cast<unsigned char>(key[key.size() - 1]);
+    return (c0 + c1 * 3 + key.size() * 17) & 0xFF;
+}
+constexpr std::size_t hd_safe_char(const char* p, std::size_t len, std::size_t idx) noexcept {
+    std::size_t has = static_cast<std::size_t>(idx < len);
+    std::size_t si = idx & (std::size_t{0} - has);
+    return static_cast<unsigned char>(p[si]) & (std::size_t{0} - has);
+}
+constexpr std::size_t hd_key_hash_2(std::string_view key) noexcept {
+    std::size_t kc = key.size();
+    kc = kc * 31 + static_cast<unsigned char>(key[0]);
+    kc = kc * 31 + hd_safe_char(key.data(), key.size(), 1);
+    return kc;
+}
+constexpr std::size_t hd_key_hash_4(std::string_view key) noexcept {
+    std::size_t kc = key.size();
+    kc = kc * 31 + static_cast<unsigned char>(key[0]);
+    kc = kc * 31 + hd_safe_char(key.data(), key.size(), 1);
+    kc = kc * 31 + hd_safe_char(key.data(), key.size(), 2);
+    kc = kc * 31 + hd_safe_char(key.data(), key.size(), 3);
+    return kc;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_hash_and_displace(
+    const std::array<std::string_view, N>& keys,
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS>& asso_values,
+    std::size_t& num_positions,
+    std::array<std::size_t, MAX_POSITIONS>& positions,
+    std::array<std::size_t, M>& slot_to_key) {
+    num_positions = select_positions<N>(keys, positions, M);
+
+    if (num_positions == 0) {
+        for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+        for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+        for (std::size_t i = 0; i < N; ++i) {
+            std::size_t slot = keys[i].size() % M;
+            if (slot_to_key[slot] != N) { return false; }
+            slot_to_key[slot] = i;
+        }
+        return true;
+    }
+
+    for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+    num_positions = HD_MODE; // sentinel for H&D mode
+    positions[0] = 0;
+    positions[1] = LAST_CHAR;
+
+    std::array<std::size_t, N> key_bucket{};
+    for (std::size_t i = 0; i < N; ++i) { key_bucket[i] = hd_bucket_hash(keys[i]); }
+
+    struct bucket_info { std::size_t ch; std::size_t count; };
+    std::array<bucket_info, N> buckets{};
+    std::size_t num_buckets = 0;
+    for (std::size_t i = 0; i < N; ++i) {
+        std::size_t bk = key_bucket[i];
+        bool found = false;
+        for (std::size_t b = 0; b < num_buckets; ++b) {
+            if (buckets[b].ch == bk) { ++buckets[b].count; found = true; break; }
+        }
+        if (!found) { buckets[num_buckets++] = {bk, 1}; }
+    }
+    for (std::size_t i = 0; i < num_buckets; ++i) {
+        for (std::size_t j = i + 1; j < num_buckets; ++j) {
+            if (buckets[j].count > buckets[i].count) {
+                auto tmp = buckets[i]; buckets[i] = buckets[j]; buckets[j] = tmp;
+            }
+        }
+    }
+
+    auto try_placement = [&](auto key_hash_fn) -> bool {
+        for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+        for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+        for (std::size_t b = 0; b < num_buckets; ++b) {
+            std::size_t ch = buckets[b].ch;
+            std::array<std::size_t, N> bucket_keys{};
+            std::size_t bk_count = 0;
+            for (std::size_t i = 0; i < N; ++i) {
+                if (key_bucket[i] == ch) { bucket_keys[bk_count++] = i; }
+            }
+            bool placed = false;
+            std::size_t max_d = M < 255 ? M : 255;
+            for (std::size_t d = 0; d < max_d; ++d) {
+                bool ok = true;
+                std::array<std::size_t, N> bucket_slots{};
+                for (std::size_t k = 0; k < bk_count; ++k) {
+                    std::size_t slot = (d + key_hash_fn(keys[bucket_keys[k]])) % M;
+                    if (slot_to_key[slot] != N) { ok = false; break; }
+                    for (std::size_t k2 = 0; k2 < k; ++k2) {
+                        if (bucket_slots[k2] == slot) { ok = false; break; }
+                    }
+                    if (!ok) { break; }
+                    bucket_slots[k] = slot;
+                }
+                if (ok) {
+                    asso_values[0][ch] = d;
+                    for (std::size_t k = 0; k < bk_count; ++k) {
+                        slot_to_key[bucket_slots[k]] = bucket_keys[k];
+                    }
+                    placed = true;
+                    break;
+                }
+            }
+            if (!placed) { return false; }
+        }
+        std::size_t filled = 0;
+        for (std::size_t i = 0; i < M; ++i) {
+            if (slot_to_key[i] != N) { ++filled; }
+        }
+        return filled == N;
+    };
+
+    if (try_placement([](std::string_view k) { return hd_key_hash_2(k); })) {
+        positions[2] = HD_HASH_2BYTE_FLAG;
+        return true;
+    }
+    if (try_placement([](std::string_view k) { return hd_key_hash_4(k); })) {
+        positions[2] = HD_HASH_4BYTE_FLAG;
+        return true;
+    }
+    return false;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_compute_phf_hd(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+    static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso{};
+    std::size_t npos{};
+    std::array<std::size_t, MAX_POSITIONS> pos{};
+    std::array<std::size_t, M> s2k{};
+    if (try_hash_and_displace<N, M>(keys, asso, npos, pos, s2k)) {
+        result.table_size = M;
+        result.asso_values = asso;
+        result.num_positions = npos;
+        result.positions = pos;
+        for (std::size_t i = 0; i < M; ++i) { result.slot_to_key[i] = s2k[i]; }
+        for (std::size_t i = M; i < phf_result<N>::MAX_TABLE_SIZE; ++i) { result.slot_to_key[i] = N; }
+        return true;
+    }
+    return false;
+}
+
+template <std::size_t N, std::size_t M>
+consteval phf_result<N> compute_phf_hd_po2(const std::array<std::string_view, N>& keys) {
+    static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+    phf_result<N> result{};
+    if (try_compute_phf_hd<N, M>(keys, result)) { return result; }
+    constexpr std::size_t NextM = M * 2;
+    if constexpr (NextM <= phf_result<N>::MAX_TABLE_SIZE) {
+        return compute_phf_hd_po2<N, NextM>(keys);
+    } else {
+        compile_time_error("Hash-and-Displace: failed to find valid table size");
+        return result;
+    }
+}
+
+// Compute a perfect hash for `keys`: try gperf at power-of-two sizes (capped so
+// the runtime tables stay within uint8 indices), then fall back to H&D.
+template <std::size_t N>
+consteval phf_result<N> compute_phf(const std::array<std::string_view, N>& keys) {
+    constexpr std::size_t StartM = next_power_of_2(N);
+    constexpr std::size_t GPERF_MAX_TABLE =
+        phf_result<N>::MAX_TABLE_SIZE < 256 ? phf_result<N>::MAX_TABLE_SIZE : 256;
+    if constexpr (StartM <= GPERF_MAX_TABLE) {
+        phf_result<N> result{};
+        if (try_gperf_po2<N, StartM, GPERF_MAX_TABLE>(keys, result)) { return result; }
+    }
+    return compute_phf_hd_po2<N, StartM>(keys);
+}
+
+// ============================================================================
+// Runtime tables (flat, uint8) derived from a phf_result.
+// ============================================================================
+
+template <std::size_t N, std::size_t TableSize, std::size_t MaxKeyLen>
+struct phf_data {
+    std::array<std::array<std::uint8_t, 256>, MAX_POSITIONS> asso_values{};
+    std::array<std::uint8_t, MAX_POSITIONS>                  positions{};
+    std::uint8_t                                             num_positions{};
+    std::uint8_t                                             hd_hash_variant{}; // 2 or 4 (H&D only)
+    std::array<std::uint8_t, TableSize>                      slot_to_key{};
+    // slot_key_bytes[s] holds the key stored at slot s, zero-padded to a 16-byte
+    // multiple so the SIMD comparison can read a whole register.
+    std::array<std::array<char, ((MaxKeyLen + 15) / 16) * 16>, TableSize> slot_key_bytes{};
+    std::array<std::uint8_t, TableSize>                      slot_key_len{};
+};
+
+template <std::size_t N>
+constexpr std::size_t compute_max_key_len(const std::array<std::string_view, N>& keys) noexcept {
+    std::size_t m = 0;
+    for (std::size_t i = 0; i < N; ++i) { if (keys[i].size() > m) { m = keys[i].size(); } }
+    return m;
+}
+
+// Validate keys and build the runtime tables from the computed perfect hash.
+template <std::size_t N, std::size_t TableSize, std::size_t MaxKeyLen>
+consteval phf_data<N, TableSize, MaxKeyLen>
+build_phf_data(const std::array<std::string_view, N>& keys, const phf_result<N>& result) {
+    for (std::size_t i = 0; i < N; ++i) {
+        if (keys[i].empty())            { compile_time_error("empty keys are not allowed in key_selector"); }
+        if (keys[i].size() > MaxKeyLen) { compile_time_error("key length exceeds MaxKeyLen"); }
+        for (char c : keys[i]) {
+            if (c == '\\') { compile_time_error("backslash not allowed in key_selector keys"); }
+            if (c == '"')  { compile_time_error("quote not allowed in key_selector keys"); }
+            if (c == '\0') { compile_time_error("null byte not allowed in key_selector keys"); }
+        }
+        for (std::size_t j = i + 1; j < N; ++j) {
+            if (keys[i] == keys[j]) { compile_time_error("duplicate keys in key_selector"); }
+        }
+    }
+
+    phf_data<N, TableSize, MaxKeyLen> out{};
+
+    if (result.num_positions == HD_MODE) {
+        // H&D mode: single displacement table in asso_values[0].
+        for (std::size_t c = 0; c < 256; ++c) {
+            out.asso_values[0][c] = static_cast<std::uint8_t>(result.asso_values[0][c]);
+        }
+        out.num_positions   = static_cast<std::uint8_t>(HD_MODE);
+        out.hd_hash_variant = static_cast<std::uint8_t>(result.positions[2]);
+    } else {
+        for (std::size_t pi = 0; pi < result.num_positions; ++pi) {
+            for (std::size_t c = 0; c < 256; ++c) {
+                out.asso_values[pi][c] = static_cast<std::uint8_t>(result.asso_values[pi][c] % TableSize);
+            }
+        }
+        out.num_positions = static_cast<std::uint8_t>(result.num_positions);
+        for (std::size_t i = 0; i < result.num_positions; ++i) {
+            out.positions[i] = (result.positions[i] == LAST_CHAR)
+                ? POS_LAST_CHAR
+                : static_cast<std::uint8_t>(result.positions[i]);
+        }
+    }
+
+    for (std::size_t s = 0; s < TableSize; ++s) {
+        out.slot_to_key[s] = static_cast<std::uint8_t>(result.slot_to_key[s]);
+    }
+
+    for (std::size_t s = 0; s < TableSize; ++s) {
+        std::size_t ki = result.slot_to_key[s];
+        if (ki < N) {
+            auto k = keys[ki];
+            out.slot_key_len[s] = static_cast<std::uint8_t>(k.size());
+            for (std::size_t c = 0; c < k.size(); ++c) { out.slot_key_bytes[s][c] = k[c]; }
+        } else {
+            out.slot_key_len[s] = 0; // empty slot: no length can match
+        }
+    }
+    return out;
+}
+
+// --- SIMD runtime primitives ------------------------------------------------
+
+// The runtime matchers read whole 16-byte blocks up to offset 63 (four blocks)
+// for keys as long as the 63-character maximum. That read must stay within the
+// buffer's trailing padding, so the padding has to exceed the largest offset we
+// touch.
+static_assert(SIMDJSON_PADDING > 63,
+              "key_selector requires SIMDJSON_PADDING > 63 for its SIMD key reads");
+
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+// True when every byte of `diff` is zero. A 32-bit-lane horizontal max is
+// enough (zero iff every 32-bit word is zero) and is cheaper than a byte-wide
+// reduction. Do not replace this with a floating-point compare against 0.0:
+// flush-to-zero would treat a denormal as zero.
+simdjson_really_inline bool neon_all_bytes_zero(uint8x16_t diff) noexcept {
+    return vmaxvq_u32(vreinterpretq_u32_u8(diff)) == 0;
+}
+#endif
+
+// Scan for the terminating '"' starting at p. Returns its byte offset (= key
+// length). Caller guarantees SIMDJSON_PADDING (== 64) bytes past the JSON buffer,
+// so reading whole 16-byte blocks up to offset 63 is always safe.
+template <std::size_t MaxKeyLen>
+simdjson_really_inline std::size_t scan_key_length(const char* p) noexcept {
+    // The closing quote of the longest key sits at most at offset MaxKeyLen, so we
+    // scan as many 16-byte blocks as it takes to cover offsets 0..MaxKeyLen. The
+    // cap (63) keeps every covered offset within the 64-byte padding guarantee, so
+    // the SIMD and scalar builds agree.
+    static_assert(MaxKeyLen <= 63, "MaxKeyLen must be <= 63 for current SIMD implementations");
+    // [[maybe_unused]]: the scalar fallback (no NEON/SSE2/LSX) does not use it.
+    [[maybe_unused]] constexpr std::size_t num_blocks = MaxKeyLen / 16 + 1;
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+    for (std::size_t b = 0; b < num_blocks; ++b) {
+        uint8x16_t v = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + b * 16);
+        uint8x16_t cmp = vceqq_u8(v, vdupq_n_u8('"'));
+        uint64_t m = vget_lane_u64(
+            vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(cmp), 4)), 0);
+        if (simdjson_likely(m != 0)) { return b * 16 + (std::size_t(__builtin_ctzll(m)) >> 2); }
+    }
+    return MaxKeyLen + 1;
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+    for (std::size_t b = 0; b < num_blocks; ++b) {
+        __m128i v   = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + b * 16));
+        __m128i cmp = _mm_cmpeq_epi8(v, _mm_set1_epi8('"'));
+        unsigned m  = static_cast<unsigned>(_mm_movemask_epi8(cmp));
+        if (simdjson_likely(m != 0)) { return b * 16 + std::size_t(__builtin_ctz(m)); }
+    }
+    return MaxKeyLen + 1;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+    for (std::size_t b = 0; b < num_blocks; ++b) {
+        __m128i v   = __lsx_vld(reinterpret_cast<const void*>(p + b * 16), 0);
+        __m128i cmp = __lsx_vseq_b(v, __lsx_vreplgr2vr_b('"'));
+        // vmskltz_b gathers the per-byte sign bits (set where the byte equals '"')
+        // into the low 16 bits of lane 0, the LSX equivalent of movemask.
+        unsigned m  = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(cmp), 0)) & 0xFFFFu;
+        if (simdjson_likely(m != 0)) { return b * 16 + std::size_t(__builtin_ctz(m)); }
+    }
+    return MaxKeyLen + 1;
+#else
+    for (std::size_t i = 0; i <= MaxKeyLen; ++i)
+        if (p[i] == '"') return i;
+    return MaxKeyLen + 1;
+#endif
+}
+
+// Byte-equal of p[0..len) against stored[0..len). stored is zero-padded past
+// `len`. Input is read over 16, 32, 48, or 64 bytes (padded JSON buffer
+// guaranteed; SIMDJSON_PADDING == 64).
+template <std::size_t MaxKeyLen>
+simdjson_really_inline bool compare_key_bytes(
+    const char* p, const char* stored, std::size_t len) noexcept {
+    // [[maybe_unused]]: only the NEON/SSE2/LSX branches read these; the scalar
+    // fallback build (no SIMD) leaves them unused, which is an error under -Werror.
+    [[maybe_unused]] alignas(16) static constexpr uint8_t idx16[16] =
+        {0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15};
+    if constexpr (MaxKeyLen <= 16) {
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+        uint8x16_t vp = vld1q_u8(reinterpret_cast<const uint8_t*>(p));
+        uint8x16_t vs = vld1q_u8(reinterpret_cast<const uint8_t*>(stored));
+        uint8x16_t mask = vcltq_u8(vld1q_u8(idx16), vdupq_n_u8(static_cast<uint8_t>(len)));
+        uint8x16_t diff = veorq_u8(vandq_u8(vp, mask), vs);
+        return neon_all_bytes_zero(diff);
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+        __m128i vp = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p));
+        __m128i vs = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored));
+        __m128i idx = _mm_load_si128(reinterpret_cast<const __m128i*>(idx16));
+        __m128i mask = _mm_cmplt_epi8(idx, _mm_set1_epi8(static_cast<char>(len)));
+        __m128i eq = _mm_cmpeq_epi8(_mm_and_si128(vp, mask), vs);
+        return _mm_movemask_epi8(eq) == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+        __m128i vp = __lsx_vld(reinterpret_cast<const void*>(p), 0);
+        __m128i vs = __lsx_vld(reinterpret_cast<const void*>(stored), 0);
+        __m128i idx = __lsx_vld(reinterpret_cast<const void*>(idx16), 0);
+        __m128i mask = __lsx_vslt_b(idx, __lsx_vreplgr2vr_b(static_cast<int>(len)));
+        __m128i eq = __lsx_vseq_b(__lsx_vand_v(vp, mask), vs);
+        return (static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq), 0)) & 0xFFFFu) == 0xFFFFu;
+#else
+        for (std::size_t i = 0; i < len; ++i)
+            if (p[i] != stored[i]) return false;
+        return true;
+#endif
+    } else if constexpr (MaxKeyLen <= 32) {
+        [[maybe_unused]] alignas(16) static constexpr uint8_t idx32_hi[16] =
+            {16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31};
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+        uint8x16_t vp_lo = vld1q_u8(reinterpret_cast<const uint8_t*>(p));
+        uint8x16_t vp_hi = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + 16);
+        uint8x16_t vs_lo = vld1q_u8(reinterpret_cast<const uint8_t*>(stored));
+        uint8x16_t vs_hi = vld1q_u8(reinterpret_cast<const uint8_t*>(stored) + 16);
+        uint8x16_t lenv = vdupq_n_u8(static_cast<uint8_t>(len));
+        uint8x16_t m_lo = vcltq_u8(vld1q_u8(idx16),    lenv);
+        uint8x16_t m_hi = vcltq_u8(vld1q_u8(idx32_hi), lenv);
+        uint8x16_t d_lo = veorq_u8(vandq_u8(vp_lo, m_lo), vs_lo);
+        uint8x16_t d_hi = veorq_u8(vandq_u8(vp_hi, m_hi), vs_hi);
+        return neon_all_bytes_zero(vorrq_u8(d_lo, d_hi));
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+        __m128i vp_lo = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p));
+        __m128i vp_hi = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + 16));
+        __m128i vs_lo = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored));
+        __m128i vs_hi = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored + 16));
+        __m128i lenv = _mm_set1_epi8(static_cast<char>(len));
+        __m128i m_lo = _mm_cmplt_epi8(_mm_load_si128(reinterpret_cast<const __m128i*>(idx16)),    lenv);
+        __m128i m_hi = _mm_cmplt_epi8(_mm_load_si128(reinterpret_cast<const __m128i*>(idx32_hi)), lenv);
+        __m128i eq_lo = _mm_cmpeq_epi8(_mm_and_si128(vp_lo, m_lo), vs_lo);
+        __m128i eq_hi = _mm_cmpeq_epi8(_mm_and_si128(vp_hi, m_hi), vs_hi);
+        return (_mm_movemask_epi8(eq_lo) & _mm_movemask_epi8(eq_hi)) == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+        __m128i lenv = __lsx_vreplgr2vr_b(static_cast<int>(len));
+        __m128i vp_lo = __lsx_vld(reinterpret_cast<const void*>(p), 0);
+        __m128i vp_hi = __lsx_vld(reinterpret_cast<const void*>(p + 16), 0);
+        __m128i vs_lo = __lsx_vld(reinterpret_cast<const void*>(stored), 0);
+        __m128i vs_hi = __lsx_vld(reinterpret_cast<const void*>(stored + 16), 0);
+        __m128i m_lo = __lsx_vslt_b(__lsx_vld(reinterpret_cast<const void*>(idx16), 0),    lenv);
+        __m128i m_hi = __lsx_vslt_b(__lsx_vld(reinterpret_cast<const void*>(idx32_hi), 0), lenv);
+        __m128i eq_lo = __lsx_vseq_b(__lsx_vand_v(vp_lo, m_lo), vs_lo);
+        __m128i eq_hi = __lsx_vseq_b(__lsx_vand_v(vp_hi, m_hi), vs_hi);
+        unsigned mlo = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq_lo), 0)) & 0xFFFFu;
+        unsigned mhi = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq_hi), 0)) & 0xFFFFu;
+        return (mlo & mhi) == 0xFFFFu;
+#else
+        for (std::size_t i = 0; i < len; ++i)
+            if (p[i] != stored[i]) return false;
+        return true;
+#endif
+    } else if constexpr (MaxKeyLen <= 64) {
+        // 3 or 4 16-byte blocks (33..48 -> 3, 49..64 -> 4). stored is zero-padded
+        // to exactly num_blocks*16 bytes (KEY_STRIDE), so neither load overruns it.
+        // [[maybe_unused]]: the scalar fallback (no NEON/SSE2/LSX) does not use it.
+        [[maybe_unused]] constexpr std::size_t num_blocks = (MaxKeyLen + 15) / 16;
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+        uint8x16_t base = vld1q_u8(idx16);
+        uint8x16_t lenv = vdupq_n_u8(static_cast<uint8_t>(len));
+        uint8x16_t acc  = vdupq_n_u8(0);
+        for (std::size_t b = 0; b < num_blocks; ++b) {
+            uint8x16_t vp   = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + b * 16);
+            uint8x16_t vs   = vld1q_u8(reinterpret_cast<const uint8_t*>(stored) + b * 16);
+            uint8x16_t idxv = vaddq_u8(base, vdupq_n_u8(static_cast<uint8_t>(b * 16)));
+            uint8x16_t mask = vcltq_u8(idxv, lenv);
+            acc = vorrq_u8(acc, veorq_u8(vandq_u8(vp, mask), vs));
+        }
+        return neon_all_bytes_zero(acc);
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+        __m128i base = _mm_load_si128(reinterpret_cast<const __m128i*>(idx16));
+        __m128i lenv = _mm_set1_epi8(static_cast<char>(len));
+        int eq = 0xFFFF;
+        for (std::size_t b = 0; b < num_blocks; ++b) {
+            __m128i vp   = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + b * 16));
+            __m128i vs   = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored + b * 16));
+            __m128i idxv = _mm_add_epi8(base, _mm_set1_epi8(static_cast<char>(b * 16)));
+            __m128i mask = _mm_cmplt_epi8(idxv, lenv);
+            eq &= _mm_movemask_epi8(_mm_cmpeq_epi8(_mm_and_si128(vp, mask), vs));
+        }
+        return eq == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+        __m128i base = __lsx_vld(reinterpret_cast<const void*>(idx16), 0);
+        __m128i lenv = __lsx_vreplgr2vr_b(static_cast<int>(len));
+        unsigned acc = 0xFFFFu;
+        for (std::size_t b = 0; b < num_blocks; ++b) {
+            __m128i vp   = __lsx_vld(reinterpret_cast<const void*>(p + b * 16), 0);
+            __m128i vs   = __lsx_vld(reinterpret_cast<const void*>(stored + b * 16), 0);
+            __m128i idxv = __lsx_vadd_b(base, __lsx_vreplgr2vr_b(static_cast<int>(b * 16)));
+            __m128i mask = __lsx_vslt_b(idxv, lenv);
+            __m128i eq   = __lsx_vseq_b(__lsx_vand_v(vp, mask), vs);
+            acc &= static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq), 0)) & 0xFFFFu;
+        }
+        return acc == 0xFFFFu;
+#else
+        for (std::size_t i = 0; i < len; ++i)
+            if (p[i] != stored[i]) return false;
+        return true;
+#endif
+    } else {
+        for (std::size_t i = 0; i < len; ++i)
+            if (p[i] != stored[i]) return false;
+        return true;
+    }
+}
+
+// --- Single 8-bit window fast path ------------------------------------------
+//
+// Many small key sets can be told apart by inspecting a *single* 8-bit window of
+// the key bytes -- and that window need not be byte-aligned. Because every JSON
+// key is terminated by a '"', the bytes at and before a key's length are well
+// defined for any key at least that long: byte i is the key character when i is
+// inside the key and the closing quote when i == len. So we read two bytes at a
+// fixed offset, extract 8 consecutive bits at a fixed intra-byte shift, and if
+// that value is distinct for every key, a 256-entry table maps it straight to a
+// candidate key. The match then needs no hash and -- in the length-free overload
+// -- no SIMD length scan: load two bytes, shift, mask, index the table, and
+// confirm the candidate with one comparison.
+//
+// Allowing an *unaligned* window (a shift of 1..7) mixes bits from two adjacent
+// bytes and discriminates key sets that no single aligned byte can. For example,
+// the partial_tweets keys {created_at,id,text,in_reply_to_status_id,user,
+// retweet_count,favorite_count} share a colliding byte at every aligned position
+// 0,1,2, yet the 8 bits starting at bit offset 2 are unique across all seven.
+// Simpler cases fall out as the shift==0 special case: {"id","screen_name"}
+// splits on byte 0, {"jo","joe"} on the quote at byte 2.
+//
+// The window is confined to the first (shortest key length + 1) bytes so the
+// two-byte read never crosses a key's closing quote into uncontrolled value
+// bytes; that final byte is the shortest key's quote.
+template <std::size_t N, std::size_t MaxKeyLen>
+struct window_data {
+    static constexpr std::size_t KEY_STRIDE = ((MaxKeyLen + 15) / 16) * 16;
+    bool                                            ok{false};
+    std::uint8_t                                    byte_offset{0}; // first byte of the 2-byte read
+    std::uint8_t                                    shift{0};       // intra-byte bit shift (0..7)
+    std::array<std::uint8_t, 256>                   window_to_key{}; // window byte -> key index, N if none
+    std::array<std::uint8_t, N>                     key_len{};
+    std::array<std::array<char, KEY_STRIDE>, N>     key_bytes{};     // zero-padded
+};
+
+// Byte seen at `idx` for key `i`: a key character when idx is inside the key, or
+// the closing '"' when idx == len (callers keep idx <= every key's length).
+template <std::size_t N>
+constexpr unsigned window_byte_at(const std::array<std::string_view, N>& keys,
+                                  std::size_t i, std::size_t idx) noexcept {
+    if (idx < keys[i].size()) { return static_cast<unsigned char>(keys[i][idx]); }
+    return static_cast<unsigned>('"');
+}
+
+// The 8-bit window value for key `i` at (byte_offset, shift): the little-endian
+// pair (byte[off], byte[off+1]) shifted right by `shift` and truncated. Matches
+// the runtime read exactly.
+template <std::size_t N>
+constexpr unsigned window_value(const std::array<std::string_view, N>& keys,
+                                std::size_t i, std::size_t byte_offset, std::size_t shift) noexcept {
+    unsigned lo = window_byte_at<N>(keys, i, byte_offset);
+    unsigned hi = window_byte_at<N>(keys, i, byte_offset + 1);
+    return ((lo | (hi << 8)) >> shift) & 0xFFu;
+}
+
+template <std::size_t N, std::size_t MaxKeyLen>
+consteval window_data<N, MaxKeyLen>
+compute_window(const std::array<std::string_view, N>& keys) {
+    window_data<N, MaxKeyLen> out{};
+
+    std::size_t min_len = keys[0].size();
+    for (std::size_t i = 1; i < N; ++i) {
+        if (keys[i].size() < min_len) { min_len = keys[i].size(); }
+    }
+
+    // Iterate windows nearest the front first (cheapest to read, smallest shift).
+    for (std::size_t off = 0; off <= min_len; ++off) {
+        for (std::size_t shift = 0; shift < 8; ++shift) {
+            // The read touches byte off, and byte off+1 when shift != 0. Both must
+            // stay within the safe region [0, min_len] (min_len is the shortest
+            // key's quote index). off <= min_len is guaranteed by the loop bound.
+            if (shift != 0 && off + 1 > min_len) { continue; }
+
+            bool distinct = true;
+            for (std::size_t i = 0; i < N && distinct; ++i) {
+                for (std::size_t j = i + 1; j < N; ++j) {
+                    if (window_value<N>(keys, i, off, shift) == window_value<N>(keys, j, off, shift)) {
+                        distinct = false;
+                        break;
+                    }
+                }
+            }
+            if (!distinct) { continue; }
+
+            out.ok          = true;
+            out.byte_offset = static_cast<std::uint8_t>(off);
+            out.shift       = static_cast<std::uint8_t>(shift);
+            for (std::size_t b = 0; b < 256; ++b) { out.window_to_key[b] = static_cast<std::uint8_t>(N); }
+            for (std::size_t i = 0; i < N; ++i) {
+                out.window_to_key[window_value<N>(keys, i, off, shift)] = static_cast<std::uint8_t>(i);
+                out.key_len[i] = static_cast<std::uint8_t>(keys[i].size());
+                for (std::size_t c = 0; c < keys[i].size(); ++c) { out.key_bytes[i][c] = keys[i][c]; }
+            }
+            return out;
+        }
+    }
+    return out; // ok == false: no single 8-bit window distinguishes the keys
+}
+
+// Read the 8-bit window at (byte_offset, shift) from a padded key pointer.
+simdjson_really_inline std::uint8_t read_window(const char* p, std::size_t byte_offset,
+                                                std::size_t shift) noexcept {
+    std::uint16_t w;
+    // Two controlled bytes (within the shortest key + its quote, hence within the
+    // padded buffer). memcpy is the portable little-endian unaligned load.
+    std::memcpy(&w, p + byte_offset, sizeof(w));
+#if defined(__BYTE_ORDER__) && (__BYTE_ORDER__ == __ORDER_BIG_ENDIAN__)
+    w = static_cast<std::uint16_t>((w >> 8) | (w << 8));
+#endif
+    return static_cast<std::uint8_t>((w >> shift) & 0xFFu);
+}
+
+// Verify a window candidate. The window table already mapped the key to the
+// single possible index `ki` (< N); here we confirm it. Folding over 0..N-1
+// turns the runtime `ki` into a compile-time index in the matching arm, so the
+// candidate's length and bytes are constants for compare_key_bytes -- the same
+// specialization the ordered find_field path gets from a CT-length
+// unsafe_is_equal, and what keeps this path competitive. The closing-quote check
+// (p[L] == '"') both confirms the key ends exactly at the candidate's length and
+// rejects a wrong-length key, so this works whether or not the caller knew len.
+template <std::size_t N, std::size_t MaxKeyLen, std::size_t... Is>
+simdjson_really_inline std::size_t
+match_window_candidate(const char* p, std::uint8_t ki,
+                       const window_data<N, MaxKeyLen>& w,
+                       std::index_sequence<Is...>) noexcept {
+  std::size_t result = N;
+  auto try_match = [&](auto Ic) {
+    constexpr std::size_t i = decltype(Ic)::value;
+    if (ki == i && p[w.key_len[i]] == '"' &&
+        key_selector_detail::compare_key_bytes<MaxKeyLen>(
+            p, w.key_bytes[i].data(), w.key_len[i])) {
+      result = i;
+    }
+  };
+  (try_match(std::integral_constant<std::size_t, Is>{}), ...);
+  return result;
+}
+
+// --- describe() string helpers (constexpr; no std::to_string, which is not) ---
+
+// Append the decimal form of v to s.
+SIMDJSON_CONSTEXPR_STRING void append_uint(std::string& s, std::size_t v) {
+    if (v == 0) { s.push_back('0'); return; }
+    char buf[20];
+    std::size_t n = 0;
+    while (v > 0) { buf[n++] = static_cast<char>('0' + (v % 10)); v /= 10; }
+    while (n > 0) { s.push_back(buf[--n]); }
+}
+
+// Append a byte as its decimal value, plus the printable character in quotes
+// when it is in the printable ASCII range (e.g. "110 ('n')").
+SIMDJSON_CONSTEXPR_STRING void append_byte(std::string& s, unsigned b) {
+    append_uint(s, b);
+    if (b >= 0x20 && b < 0x7f) {
+        s += " ('";
+        s.push_back(static_cast<char>(b));
+        s += "')";
+    }
+}
+
+} // namespace key_selector_detail
+
+/**
+ * Stateless, compile-time key selector.
+ *
+ * Usage:
+ *   using sel_t = key_selector<"id", "text", "user">;
+ *   std::size_t i = sel_t::match_raw(raw_key); // returns sel_t::size() on miss
+ *
+ * The perfect hash is built at compile time (gperf-style, with a
+ * Hash-and-Displace fallback) and only flat tables survive to runtime. All
+ * tables are static constexpr, so the lookup fully inlines.
+ *
+ * Limitations:
+ *   - Each key must be at most 63 characters long (and no longer than
+ *     SIMDJSON_PADDING). Longer keys trigger a compile-time error.
+ *   - The number of keys should be moderate. The hard limit is 255 keys;
+ *     compilation time grows with the number of keys, so prefer a few dozen at
+ *     most per selector.
+ *   - Keys must be distinct, non-empty, and free of backslash, double-quote and
+ *     null bytes (matching is done against the raw, unescaped JSON key bytes).
+ */
+template <constevalutil::fixed_string... Keys>
+struct key_selector {
+    static constexpr std::size_t N = sizeof...(Keys);
+    static_assert(N > 0,   "key_selector requires at least one key");
+    static_assert(N <= 255,"key_selector supports at most 255 keys");
+
+    static constexpr std::array<std::string_view, N> keys{ Keys.view()... };
+    static constexpr std::size_t max_key_len = key_selector_detail::compute_max_key_len<N>(keys);
+    static_assert(max_key_len <= SIMDJSON_PADDING,
+                  "key longer than SIMDJSON_PADDING is not supported");
+    // The SIMD key-length scan covers offsets 0..63 (four 16-byte blocks), which
+    // stays within the 64-byte padding guarantee. A 64-character key's closing
+    // quote would land at offset 64 and be missed on NEON/SSE2/LSX while still
+    // matching in scalar builds, so cap at 63 to keep implementations in agreement.
+    static_assert(max_key_len <= 63,
+                  "key_selector keys must be at most 63 characters long");
+
+    static constexpr auto result = key_selector_detail::compute_phf<N>(keys);
+    static constexpr std::size_t table_size = result.table_size;
+
+    static constexpr auto phf =
+        key_selector_detail::build_phf_data<N, table_size, max_key_len>(keys, result);
+
+    // Single 8-bit-window discriminator (when one exists). Detected at compile
+    // time and selected with `if constexpr` below, so the hash path is compiled
+    // out for key sets that qualify, and this is compiled out for those that do
+    // not.
+    static constexpr auto window =
+        key_selector_detail::compute_window<N, max_key_len>(keys);
+
+    static constexpr std::size_t size() noexcept { return N; }
+
+    /**
+     * Look up a JSON key whose length is already known. p must point at the first
+     * key byte (just after the opening quote) in a padded simdjson buffer, and len
+     * must be the number of raw key bytes (the distance to the closing quote).
+     * Returns the selector index in [0, N) on match, or N on miss.
+     *
+     * Prefer this overload when the caller can obtain the key length cheaply (for
+     * example, object::for_each derives it from the structural index rather than
+     * re-scanning for the closing quote).
+     */
+    static simdjson_really_inline std::size_t match_raw(const char* p, std::size_t len) noexcept {
+        if (len == 0 || len > max_key_len) { return N; }
+
+        if constexpr (window.ok) {
+            // One 8-bit window selects the only possible candidate key;
+            // match_window_candidate confirms it (bytes + closing quote). p sits
+            // in a padded buffer and the window stays within the shortest key +
+            // quote, so the two-byte read is always in bounds. len is unused here
+            // because the quote check already pins the key's end.
+            std::uint8_t ki = window.window_to_key[
+                key_selector_detail::read_window(p, window.byte_offset, window.shift)];
+            if (ki >= N) { return N; }
+            return key_selector_detail::match_window_candidate(
+                p, ki, window, std::make_index_sequence<N>{});
+        }
+
+        std::size_t slot;
+        if (phf.num_positions == key_selector_detail::HD_MODE) {
+            // Hash-and-Displace: bucket displacement + per-key hash.
+            std::string_view key(p, len);
+            std::size_t bucket = key_selector_detail::hd_bucket_hash(key);
+            std::size_t kh = (phf.hd_hash_variant == key_selector_detail::HD_HASH_2BYTE_FLAG)
+                ? key_selector_detail::hd_key_hash_2(key)
+                : key_selector_detail::hd_key_hash_4(key);
+            slot = (phf.asso_values[0][bucket] + kh) & (table_size - 1);
+        } else {
+            // gperf: h = len + sum of asso_values over the selected positions.
+            // positions / num_positions / asso_values are compile-time constants,
+            // so this loop fully unrolls. The idx < len guard mirrors the
+            // generator's char_at()-> 256 -> skip behavior for out-of-range
+            // positions (required: arbitrary positions may exceed a key's length).
+            std::size_t h = len;
+            for (std::uint8_t i = 0; i < phf.num_positions; ++i) {
+                std::uint8_t pos = phf.positions[i];
+                std::size_t idx = (pos == key_selector_detail::POS_LAST_CHAR)
+                                  ? (len - std::size_t{1})
+                                  : static_cast<std::size_t>(pos);
+                if (idx < len) {
+                    h += phf.asso_values[i][static_cast<unsigned char>(p[idx])];
+                }
+            }
+            slot = h & (table_size - 1);
+        }
+
+        std::uint8_t ki = phf.slot_to_key[slot];
+        if (ki >= N) { return N; }
+        if (phf.slot_key_len[slot] != len) { return N; }
+        if (!key_selector_detail::compare_key_bytes<max_key_len>(
+                p, phf.slot_key_bytes[slot].data(), len)) { return N; }
+        return ki;
+    }
+
+    /**
+     * Look up a JSON key. rjs must point just after an opening quote in a padded
+     * simdjson buffer. Returns the selector index in [0, N) on match, or N on miss.
+     * The key length is recovered with a SIMD scan for the closing quote; callers
+     * that already know the length should use the (p, len) overload above.
+     */
+    static simdjson_really_inline std::size_t match_raw(raw_json_string rjs) noexcept {
+        const char* p = rjs.raw();
+        if constexpr (window.ok) {
+            // One 8-bit window picks the candidate; verifying the candidate's
+            // bytes and its closing '"' confirms the full key, so the length scan
+            // is unnecessary. The window read is in bounds (padding), and the
+            // candidate length is at most max_key_len.
+            std::uint8_t ki = window.window_to_key[
+                key_selector_detail::read_window(p, window.byte_offset, window.shift)];
+            if (ki >= N) { return N; }
+            return key_selector_detail::match_window_candidate(
+                p, ki, window, std::make_index_sequence<N>{});
+        }
+        return match_raw(p, key_selector_detail::scan_key_length<max_key_len>(p));
+    }
+
+    /** Return the key text at selector index i (i in [0, N)). */
+    static constexpr std::string_view key_at(std::size_t i) noexcept {
+        return keys[i];
+    }
+
+    /**
+     * Return a complete, human-readable, multi-line description of how this
+     * selector classifies a key: which algorithm was selected at compile time
+     * (single 8-bit window, gperf-style perfect hash, or hash-and-displace), the
+     * exact bytes/positions it inspects, and the contents of the lookup tables
+     * (which window bytes or hash slots map to which key). The text mirrors what
+     * match_raw() does step by step.
+     *
+     * Everything it reports is derived from the compile-time tables, so describe()
+     * is itself usable in a constant expression when the standard library supports
+     * constexpr std::string (__cpp_lib_constexpr_string):
+     *
+     *   static_assert(!key_selector<"name", "city">::describe().empty());
+     *
+     * It allocates a std::string and is meant for documentation, debugging and
+     * tests, not for any hot path.
+     */
+    static SIMDJSON_CONSTEXPR_STRING std::string describe() {
+        std::string s;
+        s += "key_selector: ";
+        key_selector_detail::append_uint(s, N);
+        s += " keys, max key length ";
+        key_selector_detail::append_uint(s, max_key_len);
+        s += "\nkeys:\n";
+        for (std::size_t i = 0; i < N; ++i) {
+            s += "  [";
+            key_selector_detail::append_uint(s, i);
+            s += "] \"";
+            s += keys[i];
+            s += "\" (length ";
+            key_selector_detail::append_uint(s, keys[i].size());
+            s += ")\n";
+        }
+        if constexpr (window.ok) {
+            // Mirrors the window fast path of match_raw().
+            s += "algorithm: single 8-bit window\n";
+            s += "  step 1: read 2 bytes at offset ";
+            key_selector_detail::append_uint(s, window.byte_offset);
+            s += ", interpret them as a little-endian 16-bit value, shift right by ";
+            key_selector_detail::append_uint(s, window.shift);
+            s += " bits, and keep the low 8 bits\n";
+            s += "  step 2: map that byte through a 256-entry table to a key index (";
+            key_selector_detail::append_uint(s, N);
+            s += " means no match):\n";
+            for (std::size_t b = 0; b < 256; ++b) {
+                if (window.window_to_key[b] < N) {
+                    s += "    byte ";
+                    key_selector_detail::append_byte(s, static_cast<unsigned>(b));
+                    s += " -> key ";
+                    key_selector_detail::append_uint(s, window.window_to_key[b]);
+                    s += "\n";
+                }
+            }
+            s += "  step 3: confirm the candidate by checking the closing quote sits at the key's length and comparing the key bytes\n";
+        } else {
+            // Mirrors the perfect-hash path of match_raw().
+            if constexpr (phf.num_positions == key_selector_detail::HD_MODE) {
+                s += "algorithm: hash-and-displace perfect hash\n";
+                s += "  step 1: bucket = (first_byte + last_byte*3 + length*17) mod 256\n";
+                s += "  step 2: keyhash = base-31 rolling hash of the length and the first ";
+                key_selector_detail::append_uint(s, phf.hd_hash_variant);
+                s += " bytes\n";
+                s += "  step 3: slot = (displacement[bucket] + keyhash) mod ";
+                key_selector_detail::append_uint(s, table_size);
+                s += "\n  step 4: slot_to_key[slot] gives the candidate key index\n";
+                s += "  per-key derivation:\n";
+                for (std::size_t i = 0; i < N; ++i) {
+                    std::string_view k = keys[i];
+                    std::size_t bucket = key_selector_detail::hd_bucket_hash(k);
+                    std::size_t kh = (phf.hd_hash_variant == key_selector_detail::HD_HASH_2BYTE_FLAG)
+                        ? key_selector_detail::hd_key_hash_2(k)
+                        : key_selector_detail::hd_key_hash_4(k);
+                    std::size_t slot = (phf.asso_values[0][bucket] + kh) & (table_size - 1);
+                    s += "    \"";
+                    s += k;
+                    s += "\": bucket=";
+                    key_selector_detail::append_uint(s, bucket);
+                    s += " displacement=";
+                    key_selector_detail::append_uint(s, phf.asso_values[0][bucket]);
+                    s += " keyhash=";
+                    key_selector_detail::append_uint(s, kh);
+                    s += " slot=";
+                    key_selector_detail::append_uint(s, slot);
+                    s += "\n";
+                }
+            } else {
+                s += "algorithm: gperf-style perfect hash over ";
+                key_selector_detail::append_uint(s, phf.num_positions);
+                s += " character position(s)\n";
+                s += "  step 1: h = key length\n";
+                s += "  step 2: for each position below, add its association value for the key byte there (a position past the key's length contributes 0):\n";
+                for (std::size_t i = 0; i < phf.num_positions; ++i) {
+                    s += "    position ";
+                    if (phf.positions[i] == key_selector_detail::POS_LAST_CHAR) {
+                        s += "last character";
+                    } else {
+                        s += "byte index ";
+                        key_selector_detail::append_uint(s, phf.positions[i]);
+                    }
+                    s += "\n";
+                }
+                s += "  step 3: slot = h mod ";
+                key_selector_detail::append_uint(s, table_size);
+                s += " (a power of two, applied as a bitmask)\n";
+                s += "  step 4: slot_to_key[slot] gives the candidate key index\n";
+                s += "  per-key derivation:\n";
+                for (std::size_t i = 0; i < N; ++i) {
+                    std::string_view k = keys[i];
+                    std::size_t h = k.size();
+                    for (std::size_t pi = 0; pi < phf.num_positions; ++pi) {
+                        std::size_t pos = phf.positions[pi];
+                        std::size_t idx = (pos == key_selector_detail::POS_LAST_CHAR)
+                                          ? (k.size() - 1) : pos;
+                        if (idx < k.size()) {
+                            h += phf.asso_values[pi][static_cast<unsigned char>(k[idx])];
+                        }
+                    }
+                    std::size_t slot = h & (table_size - 1);
+                    s += "    \"";
+                    s += k;
+                    s += "\": h=";
+                    key_selector_detail::append_uint(s, h);
+                    s += " slot=";
+                    key_selector_detail::append_uint(s, slot);
+                    s += "\n";
+                }
+            }
+            s += "  occupied slots (slot -> key):\n";
+            for (std::size_t slot = 0; slot < table_size; ++slot) {
+                if (phf.slot_to_key[slot] < N) {
+                    s += "    slot ";
+                    key_selector_detail::append_uint(s, slot);
+                    s += " -> key ";
+                    key_selector_detail::append_uint(s, phf.slot_to_key[slot]);
+                    s += " (\"";
+                    s += keys[phf.slot_to_key[slot]];
+                    s += "\", length ";
+                    key_selector_detail::append_uint(s, phf.slot_key_len[slot]);
+                    s += ")\n";
+                }
+            }
+            s += "  confirm the candidate by checking the key length matches and comparing the key bytes\n";
+        }
+        return s;
+    }
+};
+
+namespace key_selector_detail {
+template <typename> struct is_key_selector : std::false_type {};
+template <constevalutil::fixed_string... Keys>
+struct is_key_selector<key_selector<Keys...>> : std::true_type {};
+} // namespace key_selector_detail
+
+/**
+ * Matches any instantiation of key_selector<Keys...>. Used to constrain
+ * object::for_each so that passing a non-selector type yields a clear
+ * constraint error rather than a cascade of failures inside for_each.
+ */
+template <typename T>
+concept key_selector_type = key_selector_detail::is_key_selector<T>::value;
+
+} // namespace ondemand
+} // namespace lsx
+} // namespace simdjson
+
+#endif // SIMDJSON_SUPPORTS_CONCEPTS
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+/* end file simdjson/generic/ondemand/key_selector.h for lsx */
 /* including simdjson/generic/ondemand/object.h for lsx: #include "simdjson/generic/ondemand/object.h" */
 /* begin file simdjson/generic/ondemand/object.h for lsx */
 #ifndef SIMDJSON_GENERIC_ONDEMAND_OBJECT_H
@@ -150620,6 +192416,7 @@ public:
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/implementation_simdjson_result_base.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/key_selector.h" */
 /* amalgamation skipped (editor-only): #include <vector> */
 /* amalgamation skipped (editor-only): #if SIMDJSON_STATIC_REFLECTION && SIMDJSON_SUPPORTS_CONCEPTS */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_string_builder.h"  // for constevalutil::fixed_string */
@@ -150630,6 +192427,114 @@ namespace simdjson {
 namespace lsx {
 namespace ondemand {

+#if SIMDJSON_SUPPORTS_CONCEPTS
+/**
+ * Result of object::for_each: the first error encountered (SUCCESS if none) and
+ * the number of distinct selector keys that matched during the walk. A
+ * matched_count equal to Selector::size() means every selected key was present
+ * in the object. Implicitly converts to error_code so existing callers that only
+ * care about the error (including SIMDJSON_TRY and the test ASSERT_* macros) keep
+ * working unchanged.
+ *
+ * On error paths (a handler returns a non-SUCCESS error_code, or a direct-target
+ * deserialization via value::get fails), matched_count reflects only the number
+ * of distinct selected keys that were successfully handled *before* the failure;
+ * the failing key is not counted. This is the current behavior.
+ *
+ * Marked [[nodiscard]]: for_each surfaces parse/type errors (from a matched
+ * value, a returning handler, or the structural walk itself) only through this
+ * result, so it must not be silently dropped. This matters most in builds
+ * without exceptions, where it is the only channel for those errors. To
+ * deliberately ignore it, assign to a variable or cast to void.
+ */
+struct [[nodiscard]] for_each_result {
+  error_code error{SUCCESS};
+  std::size_t matched_count{0};
+  constexpr operator error_code() const noexcept { return error; }
+};
+
+namespace key_selector_for_each_detail {
+/**
+ * A per-key handler for the variadic object::for_each is either:
+ *   - an invocable taking a value (run custom logic for that field), or
+ *   - a deserialization target T, in which case the matched value is assigned
+ *     directly via value::get(T&).
+ * The target form lets callers bind fields straight to variables without a
+ * lambda per key -- e.g. obj.for_each<"name","city","age">(name, city, age) --
+ * while still allowing a lambda in any position when custom logic is needed
+ * (for example to descend into a nested object).
+ */
+template <typename H>
+concept field_handler =
+    std::is_invocable_v<std::remove_reference_t<H>&, value> ||
+    ::simdjson::deserializable<std::remove_cvref_t<H>, value>;
+
+/**
+ * noexcept-ness of handling a single handler: nothrow-invocability for the
+ * callback form, or nothrow-deserializability for the direct-target form
+ * (builtin scalar/string targets are noexcept; custom targets follow their
+ * own tag_invoke noexcept specification).
+ */
+template <typename H>
+inline constexpr bool nothrow_field_handler_v =
+    std::is_invocable_v<std::remove_reference_t<H>&, value>
+        ? std::is_nothrow_invocable_v<std::remove_reference_t<H>&, value>
+        : ::simdjson::nothrow_deserializable<std::remove_cvref_t<H>, value>;
+
+/**
+ * nothrow_field_handler_v folded over a whole handler pack. The variadic
+ * for_each overloads spell their conditional-noexcept specifier with this
+ * single id-expression rather than an inline fold expression. MSVC (VS18)
+ * mishandles a fold-expression that appears directly in the noexcept-specifier
+ * of an out-of-line template definition: it fails to recognize the definition's
+ * specifier as matching the in-class declaration's and reports a spurious
+ * C2382 ("redefinition; different exception specifications"). Naming the fold
+ * here keeps the specifier a plain identifier, which matches reliably.
+ */
+template <typename... Handlers>
+inline constexpr bool nothrow_field_handlers_v =
+    (nothrow_field_handler_v<Handlers> && ...);
+} // namespace key_selector_for_each_detail
+#endif
+
+/**
+ * An opaque snapshot of an object's scanning position, obtained from
+ * object::get_current_position() and consumed by object::revert_position().
+ *
+ * This bundles the raw token position with the iteration depth that was
+ * live at the moment of capture: a scalar field value (e.g. a number)
+ * unwinds back to the object's own depth once fully consumed, while a
+ * compound value (array/object) not yet fully consumed stays one level
+ * deeper. The correct depth to restore to therefore depends on what was
+ * live when the snapshot was taken, not on the object itself, which is
+ * why both pieces travel together here rather than being recomputed later.
+ *
+ * The members are private and only constructible by object itself: a
+ * hand-built value here would let revert_position() jump to an arbitrary,
+ * unvalidated position and depth, and this type carries no way to check
+ * that a given snapshot still refers to a live object. Only ever pass
+ * along a value you got from get_current_position(), and only back to the
+ * same object, before that object has been reset() or the parser has
+ * iterate()d a new document: neither invalidates a snapshot in a way this
+ * type can detect.
+ */
+struct object_position {
+  /**
+   * Default-constructed so a variable can be declared and assigned later,
+   * matching e.g. document()/object(). Not a valid position to revert to.
+   */
+  simdjson_inline object_position() noexcept = default;
+
+private:
+  token_position position{};
+  depth_t depth{};
+
+  simdjson_inline object_position(token_position position_, depth_t depth_) noexcept
+    : position(position_), depth(depth_) {}
+
+  friend class object;
+};
+
 /**
  * A forward-only JSON object field iterator.
  */
@@ -150648,8 +192553,19 @@ public:
    * Using the iterator directly is also possible but error-prone and discouraged. In particular,
    * you must dereference the iterator exactly once per iteration (before calling '++').
    * Doing otherwise is unsafe and may lead to errors. You are responsible for ensuring
+   *
+   * With SIMDJSON_DEVELOPMENT_CHECKS, the iterator locks this object while it is
+   * alive, so that reentrant access (find_field(), reset(), ...) is reported as
+   * OUT_OF_ORDER_ITERATION.
+   */
+  simdjson_inline simdjson_result<object_iterator> begin() & noexcept;
+  /**
+   * Get an iterator to the start of a temporary object, e.g., `v.get_object().begin()`.
+   *
+   * The iterator does not depend on the object instance and may outlive it, so
+   * it does not lock it.
    */
-  simdjson_inline simdjson_result<object_iterator> begin() noexcept;
+  simdjson_inline simdjson_result<object_iterator> begin() && noexcept;
   simdjson_inline simdjson_result<object_iterator> end() noexcept;
   /**
    * Look up a field by name on an object (order-sensitive). By order-sensitive, we mean that
@@ -150661,10 +192577,11 @@ public:
    *
    * ```cpp
    * simdjson::ondemand::parser parser;
-   * auto obj = parser.parse(R"( { "x": 1, "y": 2, "z": 3 } )"_padded);
-   * double z = obj.find_field("z");
-   * double y = obj.find_field("y");
-   * double x = obj.find_field("x");
+   * auto json = R"( { "x": 1, "y": 2, "z": 3 } )"_padded;
+   * auto doc = parser.iterate(json);
+   * double z = doc.find_field("z");
+   * double y = doc.find_field("y");
+   * double x = doc.find_field("x");
    * ```
    * If you have multiple fields with a matching key ({"x": 1,  "x": 1}) be mindful
    * that only one field is returned.
@@ -150737,6 +192654,100 @@ public:
   /** @overload simdjson_inline simdjson_result<value> find_field_unordered(std::string_view key) & noexcept; */
   simdjson_inline simdjson_result<value> operator[](std::string_view key) && noexcept;

+#if SIMDJSON_SUPPORTS_CONCEPTS
+  /**
+   * Walk this object once and invoke on_match(selector_index, value) for each
+   * field whose key is in the compile-time key_selector Selector, in JSON order
+   * (first occurrence of a duplicate key wins). Iteration stops once all
+   * Selector::size() keys have matched or the object ends. The value is consumed
+   * in place, so this is a low-overhead way to extract a known set of fields
+   * regardless of their order in the JSON.
+   *
+   * Like other object iteration in simdjson, for_each consumes the object by
+   * advancing the underlying iterator state; after the call the same object
+   * instance should not be used for further field access or iteration.
+   *
+   * Usage:
+   *   using sel_t = ondemand::key_selector<"id", "text", "user">;
+   *   obj.for_each<sel_t>([&](std::size_t i, ondemand::value v) {
+   *     switch (i) { case 0: ...; case 1: ...; }
+   *   });
+   *
+   * Limitations (see key_selector): each key must be at most 63 characters long,
+   * and the number of keys should be moderate (hard limit 255; a handful is
+   * best, as the compile-time perfect hash may fail or slow compilation for
+   * large key sets). The keys must be distinct, non-empty, and free of backslash, double-quote and
+   * null bytes.
+   *
+   * The callback may return either void or an error_code. When it returns an
+   * error_code, the walk stops at the first non-SUCCESS result and that error is
+   * returned, which lets the callback surface value-parse errors.
+   *
+   * This function is conditionally noexcept: it is noexcept exactly when invoking
+   * the callback is noexcept. The callback runs inside this frame, so a throwing
+   * callback (e.g. one using the exception-throwing conversions like
+   * std::string_view(value) or uint64_t(value)) makes for_each potentially
+   * throwing too -- the exception propagates to the caller instead of crossing a
+   * noexcept boundary and calling std::terminate.
+   *
+   * @returns a for_each_result holding the first error encountered while walking
+   *          the object (including any error returned by the callback, SUCCESS if
+   *          none) and the number of distinct selector keys that matched. The
+   *          result converts implicitly to error_code, so callers that only need
+   *          the error can ignore the count.
+   */
+  template <typename Selector, typename Func>
+    requires key_selector_type<Selector> &&
+             std::is_invocable_v<Func&, std::size_t, value>
+  simdjson_inline for_each_result for_each(Func&& on_match)
+      noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, value>);
+
+  /**
+   * Variadic per-key form. Provide exactly one handler per key in the Selector
+   * (compiler-enforced). Handlers are processed in JSON document order for the
+   * matching keys. Each handler is either:
+   *   - a deserialization target (a variable), in which case the matched value
+   *     is assigned to it via value::get -- no lambda required; or
+   *   - an invocable taking the ondemand::value (for custom logic such as
+   *     descending into a nested object). It may return void or error_code;
+   *     returning error_code lets you surface parse/type errors.
+   * The two styles may be mixed freely, one handler per key.
+   *
+   * Example (bind fields straight to variables):
+   *   using fields = ondemand::key_selector<"name", "city", "age">;
+   *   obj.for_each<fields>(name, city, age);
+   *
+   * Example (mixing a target and a lambda):
+   *   obj.for_each<ondemand::key_selector<"id", "user">>(
+   *     id,                                          // assigned via value::get
+   *     [&](ondemand::value v){ u = read_user(v); }  // custom logic
+   *   );
+   *
+   * The index-based single-callback form (taking (size_t, value)) remains
+   * available for shared-state or more complex per-key logic.
+   */
+  template <typename Selector, typename... Handlers>
+    requires key_selector_type<Selector> &&
+             (sizeof...(Handlers) == Selector::size()) &&
+             (key_selector_for_each_detail::field_handler<Handlers> && ...)
+  simdjson_inline for_each_result for_each(Handlers&&... on_match)
+      noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
+  /**
+   * Direct-key shorthand. Equivalent to for_each<key_selector<Keys...>>(...).
+   * Lets you write the keys inline without a separate using/alias, binding each
+   * field straight to a variable (or a lambda, see the Selector form above):
+   *
+   *   obj.for_each<"name", "city", "age">(name, city, age);
+   */
+  template <constevalutil::fixed_string... Keys, typename... Handlers>
+    requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+             (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+             (key_selector_for_each_detail::field_handler<Handlers> && ...)
+  simdjson_inline for_each_result for_each(Handlers&&... on_match)
+      noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+#endif
+
   /**
    * Get the value associated with the given JSON pointer. We use the RFC 6901
    * https://tools.ietf.org/html/rfc6901 standard, interpreting the current node
@@ -150813,6 +192824,34 @@ public:
    * @returns true if the object contains some elements (not empty)
    */
   inline simdjson_result<bool> reset() & noexcept;
+  /**
+   * Get an opaque token representing the object's current scanning position.
+   * Pass it to revert_position() to return to this exact point later, without
+   * paying the cost of a full reset() and re-scan from the beginning.
+   *
+   * A typical use is an optional field that may or may not be next: capture
+   * the position, attempt find_field(), and on NO_SUCH_FIELD, revert_position()
+   * instead of reset() so that fields already consumed are not rescanned.
+   *
+   * The returned token is only valid for this object, and only until it is
+   * reset() or the parser iterate()s a new document; using it after either
+   * is undefined behavior (see object_position).
+   *
+   * @returns An opaque position token.
+   */
+  simdjson_inline object_position get_current_position() const noexcept;
+  /**
+   * Return the object's scanning position to a snapshot previously obtained
+   * from get_current_position(). Unlike reset(), this does not rescan the
+   * object from the beginning: fields before the captured position remain
+   * consumed, and scanning resumes exactly where the snapshot was captured.
+   *
+   * @param position A snapshot previously returned by get_current_position(),
+   *        for this same object.
+   * @returns SUCCESS, or OUT_OF_ORDER_ITERATION if called during active
+   *          iteration (SIMDJSON_DEVELOPMENT_CHECKS builds only).
+   */
+  simdjson_inline error_code revert_position(object_position position) noexcept;
   /**
    * This method scans the beginning of the object and checks whether the
    * object is empty.
@@ -150858,7 +192897,7 @@ public:
    */
   template <typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out)
-     noexcept(custom_deserializable<T, object> ? nothrow_custom_deserializable<T, object> : true) {
+     noexcept(nothrow_gettable<T, object>) {
     static_assert(custom_deserializable<T, object>);
     return deserialize(*this, out);
   }
@@ -150870,7 +192909,7 @@ public:
    */
   template <typename T>
   simdjson_inline simdjson_result<T> get()
-    noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+    noexcept(nothrow_gettable<T, object>)
   {
     static_assert(std::is_default_constructible<T>::value, "The specified type is not default constructible.");
     T out{};
@@ -150922,10 +192961,18 @@ protected:
   simdjson_warn_unused simdjson_inline error_code find_field_raw(const std::string_view key) noexcept;

   value_iterator iter{};
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  bool locked{false};
+  simdjson_inline void set_locked(bool _locked) noexcept;
+#endif

   friend class value;
   friend class document;
   friend struct simdjson_result<object>;
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  friend class object_iterator;
+  friend struct simdjson_result<object_iterator>;
+#endif
 };

 } // namespace ondemand
@@ -150941,7 +192988,8 @@ public:
   simdjson_inline simdjson_result(error_code error) noexcept; ///< @private
   simdjson_inline simdjson_result() noexcept = default;

-  simdjson_inline simdjson_result<lsx::ondemand::object_iterator> begin() noexcept;
+  simdjson_inline simdjson_result<lsx::ondemand::object_iterator> begin() & noexcept;
+  simdjson_inline simdjson_result<lsx::ondemand::object_iterator> begin() && noexcept;
   simdjson_inline simdjson_result<lsx::ondemand::object_iterator> end() noexcept;
   simdjson_inline simdjson_result<lsx::ondemand::value> find_field(std::string_view key) & noexcept;
   simdjson_inline simdjson_result<lsx::ondemand::value> find_field(std::string_view key) && noexcept;
@@ -150959,6 +193007,8 @@ public:
 #endif
   simdjson_inline error_code for_each_at_path_with_wildcard(std::string_view json_path, Func&& callback) noexcept;
   inline simdjson_result<bool> reset() noexcept;
+  inline simdjson_result<lsx::ondemand::object_position> get_current_position() noexcept;
+  inline error_code revert_position(lsx::ondemand::object_position position) noexcept;
   inline simdjson_result<bool> is_empty() noexcept;
   inline simdjson_result<size_t> count_fields() & noexcept;
   inline simdjson_result<std::string_view> raw_json() noexcept;
@@ -150966,7 +193016,7 @@ public:
   // TODO: move this code into object-inl.h

   template<typename T>
-  simdjson_inline simdjson_result<T> get() noexcept {
+  simdjson_inline simdjson_result<T> get() noexcept(nothrow_gettable<T, lsx::ondemand::object>) {
     if (error()) { return error(); }
     if constexpr (std::is_same_v<T, lsx::ondemand::object>) {
       return first;
@@ -150974,7 +193024,7 @@ public:
     return first.get<T>();
   }
   template<typename T>
-  simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept {
+  simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept(nothrow_gettable<T, lsx::ondemand::object>) {
     if (error()) { return error(); }
     if constexpr (std::is_same_v<T, lsx::ondemand::object>) {
       out = first;
@@ -150984,6 +193034,39 @@ public:
     return SUCCESS;
   }

+  /**
+   * Forwards to object::for_each on the underlying object, so error-code-style
+   * chains (e.g. doc["x"].get_object()) can call for_each without first
+   * extracting the object. If this result holds an error, that error is returned
+   * (with a zero match count) and the callback is not invoked. See
+   * object::for_each for the semantics.
+   */
+  template <typename Selector, typename Func>
+    requires lsx::ondemand::key_selector_type<Selector> &&
+             std::is_invocable_v<Func&, std::size_t, lsx::ondemand::value>
+  simdjson_inline lsx::ondemand::for_each_result for_each(Func&& on_match)
+      noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, lsx::ondemand::value>);
+
+  /**
+   * Forwarding overload for the variadic per-key form (targets and/or lambdas).
+   */
+  template <typename Selector, typename... Handlers>
+    requires lsx::ondemand::key_selector_type<Selector> &&
+             (sizeof...(Handlers) == Selector::size()) &&
+             (lsx::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+  simdjson_inline lsx::ondemand::for_each_result for_each(Handlers&&... on_match)
+      noexcept(lsx::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
+  /**
+   * Forwarding overload for the direct-key variadic form.
+   */
+  template <constevalutil::fixed_string... Keys, typename... Handlers>
+    requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+             (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+             (lsx::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+  simdjson_inline lsx::ondemand::for_each_result for_each(Handlers&&... on_match)
+      noexcept(lsx::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
 #if SIMDJSON_STATIC_REFLECTION
   // TODO: move this code into object-inl.h
   template<constevalutil::fixed_string... FieldNames, typename T>
@@ -151024,6 +193107,15 @@ public:
    */
   simdjson_inline object_iterator() noexcept = default;

+#if SIMDJSON_DEVELOPMENT_CHECKS
+   simdjson_inline ~object_iterator() noexcept;
+
+   simdjson_inline object_iterator(object_iterator&&) noexcept;
+   simdjson_inline object_iterator& operator=(object_iterator&&) noexcept;
+   simdjson_inline object_iterator(const object_iterator&) noexcept;
+   simdjson_inline object_iterator& operator=(const object_iterator&) noexcept;
+#endif
+
   //
   // Iterator interface
   //
@@ -151043,6 +193135,9 @@ public:
 private:
 #if SIMDJSON_DEVELOPMENT_CHECKS
    bool has_been_referenced{false};
+   object* parent{nullptr};
+
+   simdjson_inline object_iterator(const value_iterator &_iter, object* _parent) noexcept;
 #endif
   /**
    * The underlying JSON iterator.
@@ -151088,6 +193183,191 @@ public:

 #endif // SIMDJSON_GENERIC_ONDEMAND_OBJECT_ITERATOR_H
 /* end file simdjson/generic/ondemand/object_iterator.h for lsx */
+/* including simdjson/generic/ondemand/ranges.h for lsx: #include "simdjson/generic/ondemand/ranges.h" */
+/* begin file simdjson/generic/ondemand/ranges.h for lsx */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_RANGES_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_ONDEMAND_RANGES_H */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/field.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#if SIMDJSON_SUPPORTS_RANGES
+
+namespace simdjson {
+namespace lsx {
+namespace ondemand {
+
+/**
+ * A ranges-compatible iterator adapter for JSON arrays.
+ *
+ * Wraps array_iterator to satisfy std::input_iterator by providing:
+ * - const operator* (via mutable internal state)
+ * - post-increment operator
+ * - iterator_concept tag
+ *
+ * The mutable approach is standard for single-pass input iterators that
+ * read from external sources (similar to std::istream_iterator).
+ */
+class array_range_iterator {
+public:
+  using iterator_concept = std::input_iterator_tag;
+  using value_type = simdjson_result<value>;
+  using reference = simdjson_result<value>;
+  using difference_type = std::ptrdiff_t;
+
+  simdjson_inline array_range_iterator() noexcept = default;
+  simdjson_inline explicit array_range_iterator(array_iterator iter) noexcept;
+
+  /**
+   * Get the current element. Const-qualified for std::indirectly_readable;
+   * internally delegates to the mutable wrapped iterator.
+   */
+  simdjson_inline simdjson_result<value> operator*() const noexcept;
+  simdjson_inline array_range_iterator& operator++() noexcept;
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+  simdjson_inline void operator++(int) noexcept;
+SIMDJSON_POP_DISABLE_WARNINGS
+  /**
+   * Comparison delegates to array_iterator::operator==, which checks
+   * whether the underlying parser has finished the array (depth-based).
+   */
+  simdjson_inline friend bool operator==(const array_range_iterator& a,
+                                         const array_range_iterator& b) noexcept {
+    return a.iter_ == b.iter_;
+  }
+
+private:
+  mutable array_iterator iter_{};
+};
+
+/**
+ * A std::ranges::view over a JSON array.
+ *
+ * Wraps an ondemand::array and exposes begin()/end() that return
+ * array_range_iterator (satisfying std::input_iterator), enabling
+ * use with std::views::transform and other range adaptors.
+ *
+ * If the array's begin() returns an error (only possible under
+ * SIMDJSON_DEVELOPMENT_CHECKS), the range will be empty and error()
+ * will return the error code.
+ *
+ * Usage:
+ *   ondemand::parser parser;
+ *   auto doc = parser.iterate(json);
+ *   auto arr = doc.get_array().value();
+ *   for (auto elem : ondemand::get_range(arr)) { ... }
+ */
+class array_range {
+public:
+  simdjson_inline array_range() noexcept = default;
+  simdjson_inline explicit array_range(array& arr) noexcept;
+
+  simdjson_inline array_range_iterator begin() noexcept;
+  simdjson_inline array_range_iterator end() noexcept;
+
+  /** Returns SUCCESS if the range was created successfully, or the error code otherwise. */
+  simdjson_inline error_code error() const noexcept { return error_; }
+
+private:
+  array_iterator begin_{};
+  array_iterator end_{};
+  error_code error_{SUCCESS};
+};
+
+/**
+ * A ranges-compatible iterator adapter for JSON objects.
+ *
+ * Wraps object_iterator to satisfy std::input_iterator, yielding
+ * simdjson_result<field> elements (key-value pairs).
+ */
+class object_range_iterator {
+public:
+  using iterator_concept = std::input_iterator_tag;
+  using value_type = simdjson_result<field>;
+  using reference = simdjson_result<field>;
+  using difference_type = std::ptrdiff_t;
+
+  simdjson_inline object_range_iterator() noexcept = default;
+  simdjson_inline explicit object_range_iterator(object_iterator iter) noexcept;
+
+  simdjson_inline simdjson_result<field> operator*() const noexcept;
+  simdjson_inline object_range_iterator& operator++() noexcept;
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+  simdjson_inline void operator++(int) noexcept;
+SIMDJSON_POP_DISABLE_WARNINGS
+  simdjson_inline friend bool operator==(const object_range_iterator& a,
+                                         const object_range_iterator& b) noexcept {
+    return a.iter_ == b.iter_;
+  }
+
+private:
+  mutable object_iterator iter_{};
+};
+
+/**
+ * A std::ranges::view over a JSON object.
+ *
+ * Wraps an ondemand::object and exposes begin()/end() that return
+ * object_range_iterator, enabling use with range adaptors.
+ *
+ * If the object's begin() returns an error, the range will be empty
+ * and error() will return the error code.
+ */
+class object_range {
+public:
+  simdjson_inline object_range() noexcept = default;
+  simdjson_inline explicit object_range(object& obj) noexcept;
+
+  simdjson_inline object_range_iterator begin() noexcept;
+  simdjson_inline object_range_iterator end() noexcept;
+
+  /** Returns SUCCESS if the range was created successfully, or the error code otherwise. */
+  simdjson_inline error_code error() const noexcept { return error_; }
+
+private:
+  object_iterator begin_{};
+  object_iterator end_{};
+  error_code error_{SUCCESS};
+};
+
+/** Get a std::ranges compatible view over a JSON array. */
+simdjson_inline array_range get_range(array& arr) noexcept;
+
+/** Get a std::ranges compatible view over a JSON object (key-value pairs). */
+simdjson_inline object_range get_key_value_range(object& obj) noexcept;
+
+#if SIMDJSON_EXCEPTIONS
+/** Get a std::ranges compatible view, unwrapping the simdjson_result (throws on error). */
+simdjson_inline array_range get_range(simdjson_result<array> result);
+
+/** Get a std::ranges compatible view, unwrapping the simdjson_result (throws on error). */
+simdjson_inline object_range get_key_value_range(simdjson_result<object> result);
+#endif // SIMDJSON_EXCEPTIONS
+
+} // namespace ondemand
+} // namespace lsx
+} // namespace simdjson
+
+namespace std {
+namespace ranges {
+template<>
+inline constexpr bool enable_view<simdjson::lsx::ondemand::array_range> = true;
+template<>
+inline constexpr bool enable_view<simdjson::lsx::ondemand::object_range> = true;
+} // namespace ranges
+} // namespace std
+
+#endif // SIMDJSON_SUPPORTS_RANGES
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_RANGES_H
+/* end file simdjson/generic/ondemand/ranges.h for lsx */
 /* including simdjson/generic/ondemand/serialization.h for lsx: #include "simdjson/generic/ondemand/serialization.h" */
 /* begin file simdjson/generic/ondemand/serialization.h for lsx */
 #ifndef SIMDJSON_GENERIC_ONDEMAND_SERIALIZATION_H
@@ -151220,12 +193500,14 @@ inline std::ostream& operator<<(std::ostream& out, simdjson::simdjson_result<sim
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/annotations.h" */
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */

 #include <concepts>
 #include <limits>
 #if SIMDJSON_STATIC_REFLECTION
 #include <meta>
+#include <vector>
 // #include <static_reflection> // for std::define_static_string - header not available yet
 #endif

@@ -151250,10 +193532,25 @@ error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept {

 template <std::floating_point T>
 error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept {
-  double x;
-  SIMDJSON_TRY(val.get_double().get(x));
-  out = static_cast<T>(x);
-  return SUCCESS;
+  if constexpr (std::is_same_v<T, float>) {
+    // Going through binary64 and then rounding to binary32 would round twice
+    // and could produce a value that is not the float nearest to the JSON
+    // number, so we parse to binary32 directly.
+    return val.get_float().get(out);
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  } else if constexpr (std::is_same_v<T, std::float32_t>) {
+    // Same reason as float.
+    float x;
+    SIMDJSON_TRY(val.get_float().get(x));
+    out = static_cast<T>(x);
+    return SUCCESS;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+  } else {
+    double x;
+    SIMDJSON_TRY(val.get_double().get(x));
+    out = static_cast<T>(x);
+    return SUCCESS;
+  }
 }

 template <std::signed_integral T>
@@ -151289,11 +193586,59 @@ template <concepts::constructible_from_string_view T, typename ValT>
 error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(std::is_nothrow_constructible_v<T, std::string_view>) {
   std::string_view str;
   SIMDJSON_TRY(val.get_string().get(str));
-  out = T{str};
+  if constexpr (requires { out.assign(str.data(), str.size()); }) {
+    // Copy straight into out (e.g., std::string): building a temporary and
+    // move-assigning it is markedly slower.
+    out.assign(str.data(), str.size());
+  } else {
+    out = T{str};
+  }
+  return SUCCESS;
+}
+
+#if SIMDJSON_SUPPORTS_CHAR8_T
+// any C++20 char8_t string-like type (can be constructed from std::u8string_view),
+// such as std::u8string
+template <concepts::constructible_from_u8string_view T, typename ValT>
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(std::is_nothrow_constructible_v<T, std::u8string_view>) {
+  std::u8string_view str;
+  SIMDJSON_TRY(val.get_u8string().get(str));
+  if constexpr (requires { out.assign(str.data(), str.size()); }) {
+    // Copy straight into out (e.g., std::u8string), as for std::string above.
+    out.assign(str.data(), str.size());
+  } else {
+    out = T{str};
+  }
   return SUCCESS;
 }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T


+namespace details {
+// Whether to deserialize the elements of the container T directly into a new
+// element (emplace_one(out) and then get) rather than into a temporary that is
+// then moved into the container. A temporary of a trivially copyable type lives
+// in registers and is cheap to move, and value-initializing a small element in
+// place costs more than it saves (GCC zeroes it with rep stos). But moving a
+// large element, or a string (whose deserialization then needs a temporary
+// string and a move assignment of its own), is expensive.
+template <typename T>
+concept deserialize_in_place =
+    concepts::returns_reference<T> && requires(T &c) { c.pop_back(); } &&
+    !std::is_trivially_copyable_v<typename T::value_type> &&
+    (sizeof(typename T::value_type) > 32 || concepts::constructible_from_string_view<typename T::value_type>);
+
+// Removes the last element of the container on destruction while armed.
+template <typename T>
+struct pop_back_guard {
+  T &container;
+  bool armed{true};
+  ~pop_back_guard() {
+    if (armed) { container.pop_back(); }
+  }
+};
+} // namespace details
+
 /**
  * STL containers have several constructors including one that takes a single
  * size argument. Thus, some compilers (Visual Studio) will not be able to
@@ -151317,22 +193662,65 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
     SIMDJSON_TRY(val.get_array().get(arr));
   }

-  for (auto v : arr) {
-    if constexpr (concepts::returns_reference<T>) {
-      if (auto const err = v.get<value_type>().get(concepts::emplace_one(out));
-          err) {
-        // If an error occurs, the empty element that we just inserted gets
-        // removed. We're not using a temp variable because if T is a heavy
-        // type, we want the valid path to be the fast path and the slow path be
-        // the path that has errors in it.
-        if constexpr (requires { out.pop_back(); }) {
-          static_cast<void>(out.pop_back());
+  if constexpr (std::is_same_v<T, std::vector<value_type>> && !std::is_same_v<value_type, bool>) {
+    // Collect the elements in a per-thread scratch vector that keeps its
+    // capacity from call to call, then move them into out after reserving the
+    // exact size: out is allocated once instead of being regrown. A nested
+    // array of the same type finds the scratch busy and takes the paths below.
+    // Prior related work: jsonifier keeps a thread-local vector and sizes the
+    // caller's vector from that element count (parse_impl.hpp,
+    // https://github.com/nihilai-collective/Jsonifier).
+    struct scratch_space {
+      std::vector<value_type> elements{};
+      bool busy{false};
+    };
+    static thread_local scratch_space scratch;
+    if (!scratch.busy && out.empty()) {
+      struct release_scratch {
+        scratch_space &s;
+        T &out;
+        size_t parsed{0};
+        bool complete{false};
+        // On an error or an exception, out gets the elements parsed so far (as
+        // with the loops below), without allocating. Kept out of the hot path.
+        simdjson_never_inline void keep_parsed() noexcept {
+          s.elements.resize(parsed);
+          out.swap(s.elements);
         }
-        return err;
-      }
-    } else {
+        ~release_scratch() {
+          if (simdjson_unlikely(!complete)) { keep_parsed(); }
+          s.elements.clear();
+          // Do not hold on to the memory of a very large array.
+          if (s.elements.capacity() * sizeof(value_type) > (1 << 20)) { std::vector<value_type>().swap(s.elements); }
+          s.busy = false;
+        }
+      } release{scratch, out};
+      scratch.busy = true;
+      for (auto v : arr) {
+        SIMDJSON_TRY(v.get<value_type>(scratch.elements.emplace_back()));
+        release.parsed++;
+      }
+      out.reserve(release.parsed);
+      release.complete = true;
+      for (auto &e : scratch.elements) { out.emplace_back(std::move(e)); }
+      return SUCCESS;
+    }
+  }
+  if constexpr (details::deserialize_in_place<T>) {
+    for (auto v : arr) {
+      auto &slot = concepts::emplace_one(out);
+      // An error or an exception (a user tag_invoke may throw) must not leave
+      // a partially deserialized element behind.
+      details::pop_back_guard<T> guard{out};
+      SIMDJSON_TRY(v.get<value_type>(slot));
+      guard.armed = false;
+    }
+  } else {
+    for (auto v : arr) {
+      // Deserialize into a temporary first: an error or an exception (a user
+      // tag_invoke may throw) must not leave a default-constructed element behind.
       value_type temp;
-      if (auto const err = v.get<value_type>().get(temp); err) {
+      if (auto const err = v.get<value_type>(temp); err) {
         return err;
       }
       concepts::emplace_one(out, std::move(temp));
@@ -151373,7 +193761,7 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
 }

 template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, lsx::ondemand::object &obj, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, lsx::ondemand::object &obj, T &out) noexcept(false) {
   using value_type = typename std::remove_cvref_t<T>::mapped_type;

   out.clear();
@@ -151392,21 +193780,21 @@ error_code tag_invoke(deserialize_tag, lsx::ondemand::object &obj, T &out) noexc
 }

 template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, lsx::ondemand::value &val, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, lsx::ondemand::value &val, T &out) noexcept(false) {
   lsx::ondemand::object obj;
   SIMDJSON_TRY(val.get_object().get(obj));
   return simdjson::deserialize(obj, out);
 }

 template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, lsx::ondemand::document &doc, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, lsx::ondemand::document &doc, T &out) noexcept(false) {
   lsx::ondemand::object obj;
   SIMDJSON_TRY(doc.get_object().get(obj));
   return simdjson::deserialize(obj, out);
 }

 template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, lsx::ondemand::document_reference &doc, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, lsx::ondemand::document_reference &doc, T &out) noexcept(false) {
   lsx::ondemand::object obj;
   SIMDJSON_TRY(doc.get_object().get(obj));
   return simdjson::deserialize(obj, out);
@@ -151417,10 +193805,6 @@ error_code tag_invoke(deserialize_tag, lsx::ondemand::document_reference &doc, T
  * This CPO (Customization Point Object) will help deserialize into
  * smart pointers.
  *
- * If constructing T is nothrow, this conversion should be nothrow as well since
- * we return MEMALLOC if we're not able to allocate memory instead of throwing
- * the error message.
- *
  * @tparam T The type inside the smart pointer
  * @tparam ValT document/value type
  * @param val document/value
@@ -151428,7 +193812,7 @@ error_code tag_invoke(deserialize_tag, lsx::ondemand::document_reference &doc, T
  * @return status of the conversion
  */
 template <concepts::smart_pointer T, typename ValT>
-error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(nothrow_deserializable<typename std::remove_cvref_t<T>::element_type, ValT>) {
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
   using element_type = typename std::remove_cvref_t<T>::element_type;

   // For better error messages, don't use these as constraints on
@@ -151440,12 +193824,13 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(nothrow_deser
       std::is_default_constructible_v<element_type>,
       "The specified type inside the unique_ptr must default constructible.");

-  auto ptr = new (std::nothrow) element_type();
-  if (ptr == nullptr) {
+  // Own the allocation before get(): a user tag_invoke may throw.
+  std::unique_ptr<element_type> ptr(new (std::nothrow) element_type());
+  if (!ptr) {
     return MEMALLOC;
   }
   SIMDJSON_TRY(val.template get<element_type>(*ptr));
-  out.reset(ptr);
+  out = std::move(ptr);
   return SUCCESS;
 }

@@ -151477,53 +193862,595 @@ error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept(nothrow_deser

 template <typename T>
 constexpr bool user_defined_type = (std::is_class_v<T>
-&& !std::is_same_v<T, std::string> && !std::is_same_v<T, std::string_view> && !concepts::optional_type<T> &&
-!concepts::appendable_containers<T>);
+&& !std::is_same_v<T, std::string> && !std::is_same_v<T, std::string_view>
+#if SIMDJSON_SUPPORTS_CHAR8_T
+// The char8_t string types are class types with no reflectable members, so
+// without this they would be taken for user structs and the reflection
+// overload below would compete with the u8 string overload in
+// std_deserialize.h, making every tag_invoke call on them ambiguous.
+&& !std::is_same_v<T, std::u8string> && !std::is_same_v<T, std::u8string_view>
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+&& !concepts::optional_type<T> &&
+!concepts::appendable_containers<T>
+// simdjson's own types (array, object, value, raw_json_string, number,
+// document, document_reference) have dedicated get<T>() specializations and
+// must never go through reflection.
+&& !is_builtin_deserializable_v<T>
+&& !std::is_same_v<T, lsx::ondemand::number>
+&& !std::is_same_v<T, lsx::ondemand::document>
+&& !std::is_same_v<T, lsx::ondemand::document_reference>);
+
+
+// key_selector_reflection_detail is defined unconditionally (it only requires
+// static reflection). It provides the compile-time machinery for building a
+// key_selector from a struct's members, the per-member helpers shared by every
+// deserialization path (they implement the annotations of annotations.h), the
+// ordered per-member path used by the opt-out build (see
+// deserialize_struct_ordered below), and the scan used for structs annotated
+// with deny_unknown_fields and as an automatic fallback (see
+// deserialize_struct_scan and keys_fit_selector below).
+namespace key_selector_reflection_detail {
+
+// A member participates if it is public, non-const, and not annotated with skip
+// or skip_deserializing.
+consteval bool is_eligible_member(std::meta::info mem) {
+  return !std::meta::is_const(mem) && std::meta::is_public(mem)
+      && !simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_tag)
+      && !simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_deserializing_tag);
+}
+
+// A field is a data member reached from T through a chain of members: usually a
+// direct member of T (a path of length one), or a member of a member annotated
+// with flatten (the members of a flattened structure are fields of the enclosing
+// one). A field is identified by the type member_path<m1, m2, ..., leaf>.
+template <std::meta::info First, std::meta::info... Rest>
+struct member_path {
+  // The data member holding the value; its annotations drive (de)serialization.
+  static constexpr std::meta::info leaf = [] {
+    std::meta::info members[] = {First, Rest...};
+    return members[sizeof...(Rest)];
+  }();
+  template <typename T>
+  static simdjson_inline constexpr auto &get(T &obj) noexcept {
+    if constexpr (sizeof...(Rest) == 0) {
+      return obj.[:First:];
+    } else {
+      return member_path<Rest...>::get(obj.[:First:]);
+    }
+  }
+};
+
+// True when T can be flattened: a structure deserialized member by member (not
+// a string, a container, an optional, a smart pointer, ...).
+template <typename T>
+constexpr bool flattenable_type = user_defined_type<T> && !concepts::string_view_keyed_map<T>
+    && !concepts::container_but_not_string<T> && !concepts::smart_pointer<T>;
+
+consteval void append_eligible_fields(std::meta::info type, std::vector<std::meta::info> &prefix,
+                                      std::vector<std::meta::info> &fields) {
+  for (std::meta::info mem : std::meta::nonstatic_data_members_of(type, std::meta::access_context::unchecked())) {
+    if (!is_eligible_member(mem)) { continue; }
+    prefix.push_back(std::meta::reflect_constant(mem));
+    if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::flatten_tag)) {
+      std::meta::info flattened = simdjson::detail::flattened_type(mem);
+      if (!std::meta::extract<bool>(std::meta::substitute(^^flattenable_type, {flattened}))) {
+        throw std::meta::exception(u8"simdjson::flatten requires a member whose type is a structure deserialized member by member", mem);
+      }
+      append_eligible_fields(flattened, prefix, fields);
+    } else {
+      fields.push_back(std::meta::substitute(^^member_path, prefix));
+    }
+    prefix.pop_back();
+  }
+}
+
+// The fields of `type` that participate in deserialization, in declaration order
+// (the fields of a flattened member take its place). A field's position in this
+// list is its "field index" below.
+consteval std::vector<std::meta::info> eligible_fields(std::meta::info type) {
+  std::vector<std::meta::info> prefix;
+  std::vector<std::meta::info> fields;
+  append_eligible_fields(type, prefix, fields);
+  return fields;
+}
+
+// The data member at the end of a field path.
+consteval std::meta::info field_leaf(std::meta::info path) {
+  return std::meta::extract<std::meta::info>(std::meta::template_arguments_of(path).back());
+}
+
+// Number of fields that participate in deserialization. A class can have zero
+// eligible fields (e.g. std::chrono::time_point, whose only data member is
+// private): an empty key_selector cannot be built, so the tag_invoke below
+// special-cases this count.
+template <typename T>
+consteval std::size_t eligible_field_count() {
+  return eligible_fields(^^T).size();
+}
+
+// The keys accepted for a member (or an enumerator): its JSON key followed by its
+// aliases. They are static strings, so they can be used as template arguments.
+consteval std::vector<const char *> accepted_keys_of(std::meta::info entity) {
+  std::vector<const char *> keys;
+  for (std::string_view key : simdjson::detail::json_key_names(entity)) {
+    bool repeated = false;
+    for (const char *previous : keys) { repeated = repeated || std::string_view(previous) == key; }
+    if (!repeated) { keys.push_back(std::define_static_string(key)); }
+  }
+  return keys;
+}
+
+// Every key accepted for T: the eligible fields in order, each followed by its
+// aliases. This is the key list of T's key_selector.
+consteval std::vector<const char *> accepted_keys(std::meta::info type) {
+  std::vector<const char *> keys;
+  for (std::meta::info path : eligible_fields(type)) {
+    for (const char *key : accepted_keys_of(field_leaf(path))) { keys.push_back(key); }
+  }
+  return keys;
+}
+
+// The field index of each entry of accepted_keys(type).
+consteval std::vector<std::size_t> accepted_key_fields(std::meta::info type) {
+  std::vector<std::size_t> key_fields;
+  std::vector<std::meta::info> fields = eligible_fields(type);
+  for (std::size_t i = 0; i < fields.size(); ++i) {
+    for (std::size_t j = 0; j < accepted_keys_of(field_leaf(fields[i])).size(); ++j) { key_fields.push_back(i); }
+  }
+  return key_fields;
+}
+
+// True when no two eligible fields of T accept the same key. Otherwise one JSON
+// key would have to fill several members: this is reported at compile time.
+template <typename T>
+consteval bool accepted_keys_are_distinct() {
+  std::vector<const char *> keys = accepted_keys(^^T);
+  for (std::size_t i = 0; i < keys.size(); ++i) {
+    for (std::size_t j = i + 1; j < keys.size(); ++j) {
+      if (std::string_view(keys[i]) == std::string_view(keys[j])) { return false; }
+    }
+  }
+  return true;
+}
+
+// True when some accepted key of T is written with escape sequences in JSON (a
+// double quote, a backslash or a control character). Such a key can only be
+// matched by comparing unescaped keys (deserialize_struct_scan): obj[key] and
+// the key_selector compare the raw bytes.
+template <typename T>
+consteval bool keys_need_unescaping() {
+  for (std::string_view key : accepted_keys(^^T)) {
+    for (char c : key) {
+      if (c == '\\' || c == '"' || static_cast<unsigned char>(c) < 0x20) { return true; }
+    }
+  }
+  return false;
+}

+// True when some eligible field of T has aliases: several selector keys may then
+// map to the same field.
+template <typename T>
+consteval bool has_aliases() {
+  return accepted_keys(^^T).size() != eligible_fields(^^T).size();
+}
+
+// True when `mem` is annotated with default_value or default_from (directly, or
+// through a default_value annotation on the enclosing structure).
+template <auto mem>
+consteval bool has_default() {
+  return simdjson::detail::has_annotation(mem, ^^simdjson::detail::default_value_tag)
+      || simdjson::detail::has_annotation(std::meta::parent_of(mem), ^^simdjson::detail::default_value_tag)
+      || simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::default_from_t) != std::meta::info{};
+}
+
+// True when a missing key for `mem` is not an error: optional members and
+// members with a default.
+template <auto mem>
+consteval bool may_be_absent() {
+  return concepts::optional_type<typename [: std::meta::type_of(mem) :]> || has_default<mem>();
+}
+
+// True when every eligible field of T is required (none may be absent). In that
+// case presence can be checked with a single match count instead of a per-field
+// "seen" array.
+template <typename T>
+consteval bool all_eligible_fields_required() {
+  bool all_required = true;
+  template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+    if constexpr (may_be_absent<[: path :]::leaf>()) {
+      all_required = false;
+    }
+  }
+  return all_required;
+}
+
+// `key` as a constevalutil::fixed_string usable as an NTTP.
+template <const char *key>
+consteval auto key_fixed_string() {
+  constexpr std::string_view key_view{ key };
+  char buffer[key_view.size() + 1] = {};
+  for (std::size_t i = 0; i < key_view.size(); ++i) { buffer[i] = key_view[i]; }
+  return constevalutil::fixed_string<key_view.size() + 1>(buffer);
+}
+
+// key_selector template arguments (one fixed_string per accepted key), in the
+// order of accepted_keys.
+template <typename T>
+consteval std::vector<std::meta::info> selector_key_args() {
+  std::vector<std::meta::info> args;
+  template for (constexpr const char *key : std::define_static_array(accepted_keys(^^T))) {
+    args.push_back(std::meta::reflect_constant(key_fixed_string<key>()));
+  }
+  return args;
+}
+
+// key_selector whose keys are exactly T's accepted keys (index i <-> i-th key).
+template <typename T>
+using selector_for = typename [: std::meta::substitute(
+    ^^lsx::ondemand::key_selector, selector_key_args<T>()) :];
+
+// True when T's accepted keys satisfy every key_selector requirement, so a
+// key_selector can be built for T without a compile-time error. This mirrors the
+// key_selector limits (see key_selector.h): at most 255 keys, each key non-empty
+// and at most 63 characters, no backslash / double-quote / null byte, and all
+// keys distinct. Keys with any other control character are excluded too: they
+// are escaped in JSON, so their raw bytes never match. When this returns false
+// the deserializer falls back to deserialize_struct_scan instead of failing to
+// compile.
+template <typename T>
+consteval bool keys_fit_selector() {
+  std::vector<std::string_view> keys;
+  for (const char *key : accepted_keys(^^T)) { keys.push_back(key); }
+  if (keys.size() > 255) { return false; }
+  for (std::size_t i = 0; i < keys.size(); ++i) {
+    if (keys[i].empty() || keys[i].size() > 63) { return false; }
+    for (char c : keys[i]) {
+      if (c == '\\' || c == '"' || static_cast<unsigned char>(c) < 0x20) { return false; }
+    }
+    for (std::size_t j = i + 1; j < keys.size(); ++j) {
+      if (keys[i] == keys[j]) { return false; }
+    }
+  }
+  return true;
+}
+
+// True when the class `adapter` declares a member named deserialize.
+consteval bool declares_deserialize(std::meta::info adapter) {
+  for (std::meta::info m : std::meta::members_of(adapter, std::meta::access_context::unchecked())) {
+    if (std::meta::has_identifier(m) && std::meta::identifier_of(m) == "deserialize") { return true; }
+  }
+  return false;
+}
+
+// Deserialize a JSON value into `target`, the storage of member `mem`, through
+// the member's with<Adapter> annotation when the adapter provides a deserialize
+// function.
+template <auto mem, typename ValueT, typename M>
+simdjson_warn_unused simdjson_inline error_code deserialize_member_value(ValueT &field_value, M &target) noexcept(false) {
+  constexpr std::meta::info with_type = simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::with_t);
+  if constexpr (with_type != std::meta::info{}) {
+    using adapter = typename [: with_type :]::adapter;
+    using ondemand_value = lsx::ondemand::value;
+    if constexpr (requires { { adapter::deserialize(field_value, target) } -> std::convertible_to<error_code>; }) {
+      return adapter::deserialize(field_value, target);
+    } else if constexpr (requires(ondemand_value &v) { { adapter::deserialize(v, target) } -> std::convertible_to<error_code>; }
+                         && requires { field_value.get_value(); }) {
+      // A transparent structure read from a document: the adapter takes an
+      // ondemand::value. A scalar document cannot be viewed as a value, so it
+      // reports SCALAR_DOCUMENT_AS_VALUE (an adapter taking auto& receives the
+      // document itself and has no such limitation).
+      ondemand_value v;
+      SIMDJSON_TRY(field_value.get_value().get(v));
+      return adapter::deserialize(v, target);
+    } else {
+      static_assert(!declares_deserialize(^^adapter),
+                    "the deserialize function of a simdjson::with adapter must be callable as "
+                    "Adapter::deserialize(simdjson::ondemand::value &, T &) and return an error_code");
+      return field_value.get(target);
+    }
+  } else {
+    return field_value.get(target);
+  }
+}
+
+// Deserialize the storage `target` of member `mem` from a JSON value.
+template <auto mem, typename ValueT, typename M>
+simdjson_warn_unused simdjson_inline error_code deserialize_member(ValueT &field_value, M &target) noexcept(false) {
+  if constexpr (has_default<mem>() && std::is_default_constructible_v<M> && std::is_move_assignable_v<M>) {
+    // A present key replaces the default value: deserialize into a fresh
+    // temporary so that, e.g., a container does not append to its default
+    // content, and a failure leaves the default untouched.
+    M value{};
+    SIMDJSON_TRY(deserialize_member_value<mem>(field_value, value));
+    target = std::move(value);
+    return SUCCESS;
+  } else {
+    return deserialize_member_value<mem>(field_value, target);
+  }
+}

+// Deserialize the field with the given field index.
+template <typename T>
+simdjson_warn_unused simdjson_inline error_code deserialize_field_at(
+    std::size_t field_index, lsx::ondemand::value field_value, T &out) noexcept(false) {
+  std::size_t counter = 0;
+  template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+    using field = [: path :];
+    if (field_index == counter) { return deserialize_member<field::leaf>(field_value, field::get(out)); }
+    ++counter;
+  }
+  return SUCCESS;
+}
+
+// Called when the key(s) of member `mem`, stored in `target`, are missing from
+// the JSON object: an error for a required member, a call to the default_from
+// factory, or nothing at all.
+template <auto mem, typename M>
+simdjson_warn_unused simdjson_inline error_code on_missing_member(M &target) noexcept(false) {
+  constexpr std::meta::info default_from_type = simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::default_from_t);
+  if constexpr (default_from_type != std::meta::info{}) {
+    target = [: default_from_type :]::factory();
+    return SUCCESS;
+  } else if constexpr (may_be_absent<mem>()) {
+    // For optional and default_value members, a missing key is not an error:
+    // leave the member at its current (default) value.
+    (void)target;
+    return SUCCESS;
+  } else {
+    (void)target;
+    return NO_SUCH_FIELD;
+  }
+}
+
+// Report or handle every field that was not seen in the JSON object.
+template <typename T, std::size_t N>
+simdjson_warn_unused simdjson_inline error_code handle_missing_fields(
+    const std::array<bool, N> &seen_field, T &out) noexcept(false) {
+  std::size_t counter = 0;
+  template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+    using field = [: path :];
+    if (!seen_field[counter]) { SIMDJSON_TRY(on_missing_member<field::leaf>(field::get(out))); }
+    ++counter;
+  }
+  return SUCCESS;
+}
+
+// Ordered, per-field deserialization: one obj[key] lookup per eligible field
+// (and per alias, until one is found). This is the opt-out path
+// (-DSIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION=1), except for structs with keys
+// that need unescaping (see keys_need_unescaping).
+template <typename T>
+simdjson_warn_unused error_code deserialize_struct_ordered(
+    lsx::ondemand::object &obj, T &out) noexcept(false) {
+  template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+    using field = [: path :];
+    lsx::ondemand::value field_value;
+    error_code error = NO_SUCH_FIELD;
+    template for (constexpr const char *key : std::define_static_array(accepted_keys_of(field::leaf))) {
+      if (error == NO_SUCH_FIELD) { error = obj[std::string_view(key)].get(field_value); }
+    }
+    if (error == NO_SUCH_FIELD) {
+      SIMDJSON_TRY(on_missing_member<field::leaf>(field::get(out)));
+    } else if (error) {
+      return error;
+    } else {
+      SIMDJSON_TRY(deserialize_member<field::leaf>(field_value, field::get(out)));
+    }
+  }
+  return SUCCESS;
+}
+
+// Appends the JSON keys that serialization writes for members of `type` that
+// deserialization cannot assign (const or non-public members, directly or
+// through flatten). `all` is true inside a flattened member that is itself
+// unassignable. Keys of skip_deserializing members are not included: they are
+// unknown keys, as documented.
+consteval void append_unassignable_keys(std::meta::info type, bool all, std::vector<const char *> &keys) {
+  for (std::meta::info mem : std::meta::nonstatic_data_members_of(type, std::meta::access_context::unchecked())) {
+    if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_tag)
+        || simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_serializing_tag)
+        || simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_deserializing_tag)) {
+      continue;
+    }
+    bool unassignable = all || !is_eligible_member(mem);
+    if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::flatten_tag)) {
+      append_unassignable_keys(simdjson::detail::flattened_type(mem), unassignable, keys);
+    } else if (unassignable) {
+      keys.push_back(std::define_static_string(simdjson::detail::json_key_name(mem)));
+    }
+  }
+}
+
+consteval std::vector<const char *> unassignable_keys(std::meta::info type) {
+  std::vector<const char *> keys;
+  append_unassignable_keys(type, false, keys);
+  return keys;
+}
+
+// Deserialization by a single pass over every field of the object, comparing
+// unescaped keys. It is used for structs annotated with deny_unknown_fields
+// (DenyUnknown = true), where a key that does not map to an eligible field is
+// reported as UNKNOWN_FIELD, and as the fallback for structs whose keys the
+// key_selector or obj[key] cannot match (see keys_fit_selector and
+// keys_need_unescaping). As with object::for_each, the first occurrence of a
+// field wins (later duplicates, or aliases of a field already seen, are ignored).
+template <bool DenyUnknown, typename T>
+simdjson_warn_unused error_code deserialize_struct_scan(
+    lsx::ondemand::object &obj, T &out) noexcept(false) {
+  static constexpr auto keys = std::define_static_array(accepted_keys(^^T));
+  static constexpr auto key_fields = std::define_static_array(accepted_key_fields(^^T));
+  std::array<bool, eligible_field_count<T>()> seen_field{};
+  for (auto field_result : obj) {
+    lsx::ondemand::field json_field;
+    SIMDJSON_TRY(std::move(field_result).get(json_field));
+    std::string_view key;
+    SIMDJSON_TRY(json_field.unescaped_key().get(key));
+    std::size_t key_index = keys.size();
+    for (std::size_t i = 0; i < keys.size(); ++i) {
+      if (key == std::string_view(keys[i])) { key_index = i; break; }
+    }
+    if (key_index == keys.size()) {
+      if constexpr (DenyUnknown) {
+        // A key that T itself serializes (e.g. of a const member) is not
+        // unknown: a serialized value must parse back.
+        static constexpr auto ignored_keys = std::define_static_array(unassignable_keys(^^T));
+        bool ignored = false;
+        for (const char *ignored_key : ignored_keys) {
+          if (key == std::string_view(ignored_key)) { ignored = true; break; }
+        }
+        if (!ignored) { return UNKNOWN_FIELD; }
+      }
+      continue;
+    }
+    const std::size_t field_index = key_fields[key_index];
+    if (seen_field[field_index]) { continue; }
+    seen_field[field_index] = true;
+    SIMDJSON_TRY(deserialize_field_at(field_index, json_field.value(), out));
+  }
+  return handle_missing_fields(seen_field, out);
+}
+
+template <typename T>
+consteval bool is_transparent() {
+  return simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag);
+}
+
+} // namespace key_selector_reflection_detail
+
+// Deserialize a reflected struct. By default this builds a compile-time
+// key_selector from the struct's members and walks each object once with
+// object::for_each (perfect-hash key matching), instead of one obj[key] lookup
+// per member. Other paths are used instead:
+//   - globally, the ordered per-member path when defining
+//     -DSIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION=1;
+//   - automatically and per-type, a scan of the object comparing unescaped keys
+//     when the struct's keys do not fit the key_selector limits (see
+//     keys_fit_selector) or cannot be compared raw (see keys_need_unescaping),
+//     so that long member names and the like keep compiling rather than
+//     tripping a static_assert.
+// Structs annotated with deny_unknown_fields always use the strict scan, and
+// structs annotated with transparent are deserialized as their single member.
+//
+// noexcept(false): a member's tag_invoke may throw and the exception must reach
+// the caller of get<T>().
 template <typename T, typename ValT>
   requires(user_defined_type<T> && std::is_class_v<T>)
-error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
+  if constexpr (key_selector_reflection_detail::is_transparent<T>()) {
+    constexpr auto mem = simdjson::detail::transparent_member(^^T);
+    if constexpr (std::is_same_v<std::remove_cvref_t<ValT>, lsx::ondemand::object>) {
+      // We were handed an object: only a structure can be deserialized from it.
+      if constexpr (user_defined_type<typename [: std::meta::type_of(mem) :]>) {
+        return tag_invoke(deserialize_tag{}, val, out.[:mem:]);
+      } else {
+        return INCORRECT_TYPE;
+      }
+    } else {
+      return key_selector_reflection_detail::deserialize_member<mem>(val, out.[:mem:]);
+    }
+  } else {
+  static_assert(key_selector_reflection_detail::accepted_keys_are_distinct<T>(),
+                "two members of this structure accept the same JSON key (check rename, alias, "
+                "rename_all and flatten)");
   lsx::ondemand::object obj;
   if constexpr (std::is_same_v<std::remove_cvref_t<ValT>, lsx::ondemand::object>) {
     obj = val;
   } else {
     SIMDJSON_TRY(val.get_object().get(obj));
   }
-  template for (constexpr auto mem : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
-    if constexpr (!std::meta::is_const(mem) && std::meta::is_public(mem)) {
-      constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
-      if constexpr (concepts::optional_type<decltype(out.[:mem:])>) {
-        // for optional members, it's ok if the key is missing
-        auto error = obj[key].get(out.[:mem:]);
-        if (error && error != NO_SUCH_FIELD) {
-          if(error == NO_SUCH_FIELD) {
-            out.[:mem:].reset();
-            continue;
-          }
-          return error;
-        }
-      } else {
-        // for non-optional members, the key must be present
-        SIMDJSON_TRY(obj[key].get(out.[:mem:]));
+  if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::deny_unknown_fields_tag)) {
+    return key_selector_reflection_detail::deserialize_struct_scan<true>(obj, out);
+  } else {
+#if defined(SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION) && SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION
+  // Opt-out build: use the ordered per-member path, unless obj[key] cannot
+  // match T's keys.
+  if constexpr (key_selector_reflection_detail::keys_need_unescaping<T>()) {
+    return key_selector_reflection_detail::deserialize_struct_scan<false>(obj, out);
+  } else {
+    return key_selector_reflection_detail::deserialize_struct_ordered(obj, out);
+  }
+#else
+  if constexpr (key_selector_reflection_detail::eligible_field_count<T>() == 0) {
+    // No fields to deserialize: an empty key_selector cannot be built, so just
+    // validate that the input is an object (done above) and succeed. Mirrors the
+    // ordered per-member path, which iterates over zero members.
+    (void)out;
+    (void)obj;
+    return SUCCESS;
+  } else if constexpr (!key_selector_reflection_detail::keys_fit_selector<T>()) {
+    // Automatic fallback: T's accepted keys do not fit the key_selector limits
+    // (e.g. a member name longer than 63 characters, or a key with a double
+    // quote), so building a selector would be a compile error. Scan the object
+    // instead, so the default never breaks a struct that the opt-out path would
+    // accept.
+    return key_selector_reflection_detail::deserialize_struct_scan<false>(obj, out);
+  } else {
+  using selector = key_selector_reflection_detail::selector_for<T>;
+  if constexpr (key_selector_reflection_detail::all_eligible_fields_required<T>()
+                && !key_selector_reflection_detail::has_aliases<T>()) {
+    // Fast path: every member is required and has a single key. A single
+    // for_each pass parses each matched field; the returned match count then
+    // tells us whether every member was present (matched_count ==
+    // selector::size()) without a per-member "seen" array. A value-parse error
+    // (e.g. a type mismatch) is propagated by for_each.
+    auto walk = obj.template for_each<selector>(
+        [&](std::size_t matched_index, lsx::ondemand::value field_value) -> error_code {
+      std::size_t counter = 0;
+      template for (constexpr auto path : std::define_static_array(key_selector_reflection_detail::eligible_fields(^^T))) {
+        using field = [: path :];
+        if (matched_index == counter) { return key_selector_reflection_detail::deserialize_member<field::leaf>(field_value, field::get(out)); }
+        ++counter;
       }
-    }
-  };
-  return simdjson::SUCCESS;
-}
-
-// Support for enum deserialization - deserialize from string representation using expand approach from P2996R12
+      return SUCCESS;
+    });
+    if (walk.error) { return walk.error; }
+    // A missing required member shows up as a short match count and is reported as
+    // NO_SUCH_FIELD, mirroring the ordered obj[key] path.
+    if (walk.matched_count != selector::size()) { return NO_SUCH_FIELD; }
+    return SUCCESS;
+  } else {
+    static constexpr auto key_fields = std::define_static_array(key_selector_reflection_detail::accepted_key_fields(^^T));
+    std::array<bool, key_selector_reflection_detail::eligible_field_count<T>()> seen_field{};
+    // Single pass over the object: each field whose key matches a member (or one
+    // of its aliases) yields its selector index, which we map back to the
+    // corresponding member. The first key seen for a member wins. The callback
+    // returns an error_code so that a value-parse error (e.g. a type mismatch on
+    // a matched field) is propagated by for_each instead of being silently dropped.
+    error_code walk_error = obj.template for_each<selector>(
+        [&](std::size_t matched_index, lsx::ondemand::value field_value) -> error_code {
+      const std::size_t field_index = key_fields[matched_index];
+      if (seen_field[field_index]) { return SUCCESS; }
+      seen_field[field_index] = true;
+      return key_selector_reflection_detail::deserialize_field_at(field_index, field_value, out);
+    });
+    if (walk_error) { return walk_error; }
+    // Required members must be present: a missing one is reported as
+    // NO_SUCH_FIELD, mirroring the ordered obj[key] path. Optional and defaulted
+    // members may be absent.
+    return key_selector_reflection_detail::handle_missing_fields(seen_field, out);
+  }
+  }
+#endif // SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION
+  }
+  }
+}
+
+// Support for enum deserialization - deserialize from string representation using
+// expand approach from P2996R12. The accepted strings are the enumerator's JSON key
+// (see rename and rename_all) and its aliases.
 template <typename T, typename ValT>
   requires(std::is_enum_v<T>)
 error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
 #if SIMDJSON_STATIC_REFLECTION
   std::string_view str;
   SIMDJSON_TRY(val.get_string().get(str));
-  constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
+  static constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
   template for (constexpr auto enum_val : enumerators) {
-    if (str == std::meta::identifier_of(enum_val)) {
-      out = [:enum_val:];
-      return SUCCESS;
+    template for (constexpr const char *key : std::define_static_array(key_selector_reflection_detail::accepted_keys_of(enum_val))) {
+      if (str == std::string_view(key)) {
+        out = [:enum_val:];
+        return SUCCESS;
+      }
     }
   };

@@ -151539,33 +194466,25 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {

 template <typename simdjson_value, typename T>
   requires(user_defined_type<std::remove_cvref_t<T>>)
-error_code tag_invoke(deserialize_tag, simdjson_value &val, std::unique_ptr<T> &out) noexcept {
-  if (!out) {
-    out = std::make_unique<T>();
-    if (!out) {
-      return MEMALLOC;
-    }
-  }
-  if (auto err = val.get(*out)) {
-    out.reset();
-    return err;
+error_code tag_invoke(deserialize_tag, simdjson_value &val, std::unique_ptr<T> &out) noexcept(false) {
+  std::unique_ptr<T> ptr(new (std::nothrow) T());
+  if (!ptr) {
+    return MEMALLOC;
   }
+  SIMDJSON_TRY(val.get(*ptr));
+  out = std::move(ptr);
   return SUCCESS;
 }

 template <typename simdjson_value, typename T>
   requires(user_defined_type<std::remove_cvref_t<T>>)
-error_code tag_invoke(deserialize_tag, simdjson_value &val, std::shared_ptr<T> &out) noexcept {
-  if (!out) {
-    out = std::make_shared<T>();
-    if (!out) {
-      return MEMALLOC;
-    }
-  }
-  if (auto err = val.get(*out)) {
-    out.reset();
-    return err;
+error_code tag_invoke(deserialize_tag, simdjson_value &val, std::shared_ptr<T> &out) noexcept(false) {
+  std::shared_ptr<T> ptr(new (std::nothrow) T());
+  if (!ptr) {
+    return MEMALLOC;
   }
+  SIMDJSON_TRY(val.get(*ptr));
+  out = std::move(ptr);
   return SUCCESS;
 }

@@ -151877,9 +194796,17 @@ simdjson_inline simdjson_result<array> array::started(value_iterator &iter) noex
   return array(iter);
 }

-simdjson_inline simdjson_result<array_iterator> array::begin() noexcept {
+simdjson_inline simdjson_result<array_iterator> array::begin() & noexcept {
 #if SIMDJSON_DEVELOPMENT_CHECKS
-  if (!iter.is_at_iterator_start()) { return OUT_OF_ORDER_ITERATION; }
+  if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
+  return array_iterator(iter, this);
+#endif
+  return array_iterator(iter);
+}
+simdjson_inline simdjson_result<array_iterator> array::begin() && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  // The array is a temporary that the iterator may outlive: do not lock it.
+  if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
 #endif
   return array_iterator(iter);
 }
@@ -151906,6 +194833,9 @@ simdjson_inline simdjson_result<std::string_view> array::raw_json() noexcept {
 SIMDJSON_PUSH_DISABLE_WARNINGS
 SIMDJSON_DISABLE_STRICT_OVERFLOW_WARNING
 simdjson_inline simdjson_result<size_t> array::count_elements() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   size_t count{0};
   // Important: we do not consume any of the values.
   for(simdjson_unused auto v : *this) { count++; }
@@ -151919,6 +194849,9 @@ simdjson_inline simdjson_result<size_t> array::count_elements() & noexcept {
 SIMDJSON_POP_DISABLE_WARNINGS

 simdjson_inline simdjson_result<bool> array::is_empty() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   bool is_not_empty;
   auto error = iter.reset_array().get(is_not_empty);
   if(error) { return error; }
@@ -151926,31 +194859,30 @@ simdjson_inline simdjson_result<bool> array::is_empty() & noexcept {
 }

 inline simdjson_result<bool> array::reset() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   return iter.reset_array();
 }

+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline void array::set_locked(bool _locked) noexcept {
+  locked = _locked;
+}
+#endif
+
 inline simdjson_result<value> array::at_pointer(std::string_view json_pointer) noexcept {
-  if (json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
+  // An empty pointer has no json_pointer[0]: with a default-constructed
+  // std::string_view, reading it dereferences a null pointer.
+  if (json_pointer.empty() || json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
   json_pointer = json_pointer.substr(1);
   // - means "the append position" or "the element after the end of the array"
   // We don't support this, because we're returning a real element, not a position.
   if (json_pointer == "-") { return INDEX_OUT_OF_BOUNDS; }

-  // Read the array index
   size_t array_index = 0;
   size_t i;
-  for (i = 0; i < json_pointer.length() && json_pointer[i] != '/'; i++) {
-    uint8_t digit = uint8_t(json_pointer[i] - '0');
-    // Check for non-digit in array index. If it's there, we're trying to get a field in an object
-    if (digit > 9) { return INCORRECT_TYPE; }
-    array_index = array_index*10 + digit;
-  }
-
-  // 0 followed by other digits is invalid
-  if (i > 1 && json_pointer[0] == '0') { return INVALID_JSON_POINTER; } // "JSON pointer array index has other characters after 0"
-
-  // Empty string is invalid; so is a "/" with no digits before it
-  if (i == 0) { return INVALID_JSON_POINTER; } // "Empty string in JSON pointer array index"
+  SIMDJSON_TRY(internal::parse_json_pointer_array_index(json_pointer, array_index, i));
   // Get the child
   auto child = at(array_index);
   // If there is an error, it ends here
@@ -152024,6 +194956,9 @@ inline error_code array::for_each_at_path_with_wildcard(std::string_view json_pa
 }

 simdjson_inline simdjson_result<value> array::at(size_t index) noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   size_t i = 0;
   for (auto value : *this) {
     if (i == index) { return value; }
@@ -152053,10 +194988,14 @@ simdjson_inline simdjson_result<lsx::ondemand::array>::simdjson_result(
 {
 }

-simdjson_inline simdjson_result<lsx::ondemand::array_iterator> simdjson_result<lsx::ondemand::array>::begin() noexcept {
+simdjson_inline simdjson_result<lsx::ondemand::array_iterator> simdjson_result<lsx::ondemand::array>::begin() & noexcept {
   if (error()) { return error(); }
   return first.begin();
 }
+simdjson_inline simdjson_result<lsx::ondemand::array_iterator> simdjson_result<lsx::ondemand::array>::begin() && noexcept {
+  if (error()) { return error(); }
+  return std::move(first).begin();
+}
 simdjson_inline simdjson_result<lsx::ondemand::array_iterator> simdjson_result<lsx::ondemand::array>::end() noexcept {
   if (error()) { return error(); }
   return first.end();
@@ -152119,6 +195058,59 @@ simdjson_inline array_iterator::array_iterator(const value_iterator &_iter) noex
   : iter{_iter}
 {}

+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline array_iterator::array_iterator(const value_iterator &_iter, array* _parent) noexcept
+  : parent{_parent}, iter{_iter}
+{
+  if (parent) parent->set_locked(true);
+}
+
+simdjson_inline array_iterator::~array_iterator() noexcept
+{
+  if (parent) parent->set_locked(false);
+}
+
+simdjson_inline array_iterator::array_iterator(array_iterator&& other) noexcept
+  : has_been_referenced{other.has_been_referenced},
+    parent{other.parent},
+    iter{std::move(other.iter)}
+{
+  other.parent = nullptr;
+}
+
+simdjson_inline array_iterator& array_iterator::operator=(array_iterator&& other) noexcept {
+  if (this != &other)
+  {
+    if (parent)
+      parent->set_locked(false);
+    has_been_referenced = other.has_been_referenced;
+    parent = other.parent;
+    iter = std::move(other.iter);
+
+    other.parent = nullptr;
+  }
+  return *this;
+}
+
+simdjson_inline array_iterator::array_iterator(const array_iterator& other) noexcept
+  : has_been_referenced{other.has_been_referenced},
+    parent{nullptr},
+    iter{other.iter}
+{}
+
+simdjson_inline array_iterator& array_iterator::operator=(const array_iterator& other) noexcept {
+  if (this != &other)
+  {
+    if (parent)
+      parent->set_locked(false);
+    has_been_referenced = other.has_been_referenced;
+    parent = nullptr;
+    iter = other.iter;
+  }
+  return *this;
+}
+#endif
+
 simdjson_inline simdjson_result<value> array_iterator::operator*() noexcept {
 #if SIMDJSON_DEVELOPMENT_CHECKS
    SIMDJSON_ASSUME(!has_been_referenced);
@@ -152214,6 +195206,41 @@ namespace simdjson {
 namespace lsx {
 namespace ondemand {

+/**
+ * @private Narrows the result of get_uint64() to a smaller unsigned integer type T.
+ * @returns NUMBER_OUT_OF_RANGE if the value does not fit in a T.
+ */
+template <typename T>
+simdjson_inline simdjson_result<T> narrow_integer(simdjson_result<uint64_t> wide) noexcept {
+  uint64_t result;
+  SIMDJSON_TRY(std::move(wide).get(result));
+  if (result > static_cast<uint64_t>((std::numeric_limits<T>::max)())) { return NUMBER_OUT_OF_RANGE; }
+  return static_cast<T>(result);
+}
+
+/**
+ * @private Narrows the result of get_int64() to a smaller signed integer type T.
+ * @returns NUMBER_OUT_OF_RANGE if the value does not fit in a T.
+ */
+template <typename T>
+simdjson_inline simdjson_result<T> narrow_integer(simdjson_result<int64_t> wide) noexcept {
+  int64_t result;
+  SIMDJSON_TRY(std::move(wide).get(result));
+  if (result > static_cast<int64_t>((std::numeric_limits<T>::max)()) || result < static_cast<int64_t>((std::numeric_limits<T>::min)())) { return NUMBER_OUT_OF_RANGE; }
+  return static_cast<T>(result);
+}
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+// get_float32() parses with get_float(), so float must be binary32.
+static_assert(std::numeric_limits<float>::is_iec559 && std::numeric_limits<float>::digits == 24,
+              "get_float32() requires float to be IEEE binary32");
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+// get_float64() parses with get_double(), so double must be binary64.
+static_assert(std::numeric_limits<double>::is_iec559 && std::numeric_limits<double>::digits == 53,
+              "get_float64() requires double to be IEEE binary64");
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
+
 simdjson_inline value::value(const value_iterator &_iter) noexcept
   : iter{_iter}
 {
@@ -152245,6 +195272,13 @@ simdjson_inline simdjson_result<raw_json_string> value::get_raw_json_string() no
 simdjson_inline simdjson_result<std::string_view> value::get_string(bool allow_replacement) noexcept {
   return iter.get_string(allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> value::get_u8string(bool allow_replacement) noexcept {
+  std::string_view content;
+  SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+  return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code value::get_string(string_type& receiver, bool allow_replacement) noexcept {
   return iter.get_string(receiver, allow_replacement);
@@ -152258,6 +195292,12 @@ simdjson_inline simdjson_result<double> value::get_double() noexcept {
 simdjson_inline simdjson_result<double> value::get_double_in_string() noexcept {
   return iter.get_double_in_string();
 }
+simdjson_inline simdjson_result<float> value::get_float() noexcept {
+  return iter.get_float();
+}
+simdjson_inline simdjson_result<float> value::get_float_in_string() noexcept {
+  return iter.get_float_in_string();
+}
 simdjson_inline simdjson_result<uint64_t> value::get_uint64() noexcept {
   return iter.get_uint64();
 }
@@ -152271,17 +195311,37 @@ simdjson_inline simdjson_result<int64_t> value::get_int64_in_string() noexcept {
   return iter.get_int64_in_string();
 }
 simdjson_inline simdjson_result<uint32_t> value::get_uint32() noexcept {
-  uint64_t result;
-  SIMDJSON_TRY(get_uint64().get(result));
-  if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<uint32_t>(result);
+  return narrow_integer<uint32_t>(get_uint64());
 }
 simdjson_inline simdjson_result<int32_t> value::get_int32() noexcept {
-  int64_t result;
-  SIMDJSON_TRY(get_int64().get(result));
-  if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<int32_t>(result);
+  return narrow_integer<int32_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint16_t> value::get_uint16() noexcept {
+  return narrow_integer<uint16_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int16_t> value::get_int16() noexcept {
+  return narrow_integer<int16_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint8_t> value::get_uint8() noexcept {
+  return narrow_integer<uint8_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int8_t> value::get_int8() noexcept {
+  return narrow_integer<int8_t>(get_int64());
 }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> value::get_float32() noexcept {
+  float result;
+  SIMDJSON_TRY(get_float().get(result));
+  return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> value::get_float64() noexcept {
+  double result;
+  SIMDJSON_TRY(get_double().get(result));
+  return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<bool> value::get_bool() noexcept {
   return iter.get_bool();
 }
@@ -152293,12 +195353,26 @@ template<> simdjson_inline simdjson_result<array> value::get() noexcept { return
 template<> simdjson_inline simdjson_result<object> value::get() noexcept { return get_object(); }
 template<> simdjson_inline simdjson_result<raw_json_string> value::get() noexcept { return get_raw_json_string(); }
 template<> simdjson_inline simdjson_result<std::string_view> value::get() noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> value::get() noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_inline simdjson_result<number> value::get() noexcept { return get_number(); }
 template<> simdjson_inline simdjson_result<double> value::get() noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> value::get() noexcept { return get_float(); }
 template<> simdjson_inline simdjson_result<uint64_t> value::get() noexcept { return get_uint64(); }
 template<> simdjson_inline simdjson_result<int64_t> value::get() noexcept { return get_int64(); }
 template<> simdjson_inline simdjson_result<uint32_t> value::get() noexcept { return get_uint32(); }
 template<> simdjson_inline simdjson_result<int32_t> value::get() noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> value::get() noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> value::get() noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> value::get() noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> value::get() noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> value::get() noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> value::get() noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<> simdjson_inline simdjson_result<bool> value::get() noexcept { return get_bool(); }


@@ -152306,12 +195380,26 @@ template<> simdjson_warn_unused simdjson_inline error_code value::get(array& out
 template<> simdjson_warn_unused simdjson_inline error_code value::get(object& out) noexcept { return get_object().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(raw_json_string& out) noexcept { return get_raw_json_string().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(std::string_view& out) noexcept { return get_string(false).get(out); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::u8string_view& out) noexcept { return get_u8string(false).get(out); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_warn_unused simdjson_inline error_code value::get(number& out) noexcept { return get_number().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(double& out) noexcept { return get_double().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(float& out) noexcept { return get_float().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(uint64_t& out) noexcept { return get_uint64().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(int64_t& out) noexcept { return get_int64().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(uint32_t& out) noexcept { return get_uint32().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(int32_t& out) noexcept { return get_int32().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(uint16_t& out) noexcept { return get_uint16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(int16_t& out) noexcept { return get_int16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(uint8_t& out) noexcept { return get_uint8().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(int8_t& out) noexcept { return get_int8().get(out); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::float32_t& out) noexcept { return get_float32().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::float64_t& out) noexcept { return get_float64().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<>  simdjson_warn_unused simdjson_inline error_code value::get(bool& out) noexcept { return get_bool().get(out); }

 #if SIMDJSON_EXCEPTIONS
@@ -152480,6 +195568,9 @@ inline bool is_pointer_well_formed(std::string_view json_pointer) noexcept {
 }

 simdjson_inline simdjson_result<value> value::at_pointer(std::string_view json_pointer) noexcept {
+  // The empty JSON Pointer refers to the whole value (RFC 6901), as in
+  // document::at_pointer.
+  if (json_pointer.empty()) { return value(iter); }
   json_type t;
   SIMDJSON_TRY(type().get(t));
   switch (t)
@@ -152517,6 +195608,10 @@ template <typename Func>
 template <typename Func>
 #endif
 inline error_code value::for_each_at_path_with_wildcard(std::string_view json_path, Func&& callback) noexcept {
+  // Every recursive step of for_each_at_path_with_wildcard goes through this
+  // function, and each one descends one level into the document. A path with
+  // many segments applied to a deeply nested document would otherwise recurse
+  // without bound and overflow the stack.
   if (size_t(iter.depth()) >= iter.json_iter().parser->max_depth()) { return DEPTH_ERROR; }
   json_type t;
   SIMDJSON_TRY(type().get(t));
@@ -152630,10 +195725,46 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<lsx::ondemand::value>::
   if (error()) { return error(); }
   return first.get_int32();
 }
+simdjson_inline simdjson_result<uint16_t> simdjson_result<lsx::ondemand::value>::get_uint16() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<lsx::ondemand::value>::get_int16() noexcept {
+  if (error()) { return error(); }
+  return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<lsx::ondemand::value>::get_uint8() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<lsx::ondemand::value>::get_int8() noexcept {
+  if (error()) { return error(); }
+  return first.get_int8();
+}
 simdjson_inline simdjson_result<double> simdjson_result<lsx::ondemand::value>::get_double() noexcept {
   if (error()) { return error(); }
   return first.get_double();
 }
+simdjson_inline simdjson_result<float> simdjson_result<lsx::ondemand::value>::get_float() noexcept {
+  if (error()) { return error(); }
+  return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<lsx::ondemand::value>::get_float_in_string() noexcept {
+  if (error()) { return error(); }
+  return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<lsx::ondemand::value>::get_float32() noexcept {
+  if (error()) { return error(); }
+  return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<lsx::ondemand::value>::get_float64() noexcept {
+  if (error()) { return error(); }
+  return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<double> simdjson_result<lsx::ondemand::value>::get_double_in_string() noexcept {
   if (error()) { return error(); }
   return first.get_double_in_string();
@@ -152642,6 +195773,12 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<lsx::ondemand:
   if (error()) { return error(); }
   return first.get_string(allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<lsx::ondemand::value>::get_u8string(bool allow_replacement) noexcept {
+  if (error()) { return error(); }
+  return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_inline error_code simdjson_result<lsx::ondemand::value>::get_string(string_type& receiver, bool allow_replacement) noexcept {
   if (error()) { return error(); }
@@ -152670,11 +195807,23 @@ template<> simdjson_inline error_code simdjson_result<lsx::ondemand::value>::get
   return SUCCESS;
 }

-template<typename T> simdjson_inline simdjson_result<T> simdjson_result<lsx::ondemand::value>::get() noexcept {
+template<typename T> simdjson_inline simdjson_result<T> simdjson_result<lsx::ondemand::value>::get()
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lsx::ondemand::value>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>();
 }
-template<typename T> simdjson_inline error_code simdjson_result<lsx::ondemand::value>::get(T &out) noexcept {
+template<typename T> simdjson_inline error_code simdjson_result<lsx::ondemand::value>::get(T &out)
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lsx::ondemand::value>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>(out);
 }
@@ -152944,16 +196093,22 @@ simdjson_inline simdjson_result<int64_t> document::get_int64_in_string() noexcep
   return get_root_value_iterator().get_root_int64_in_string(true);
 }
 simdjson_inline simdjson_result<uint32_t> document::get_uint32() noexcept {
-  uint64_t result;
-  SIMDJSON_TRY(get_uint64().get(result));
-  if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<uint32_t>(result);
+  return narrow_integer<uint32_t>(get_uint64());
 }
 simdjson_inline simdjson_result<int32_t> document::get_int32() noexcept {
-  int64_t result;
-  SIMDJSON_TRY(get_int64().get(result));
-  if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<int32_t>(result);
+  return narrow_integer<int32_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint16_t> document::get_uint16() noexcept {
+  return narrow_integer<uint16_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int16_t> document::get_int16() noexcept {
+  return narrow_integer<int16_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint8_t> document::get_uint8() noexcept {
+  return narrow_integer<uint8_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int8_t> document::get_int8() noexcept {
+  return narrow_integer<int8_t>(get_int64());
 }
 simdjson_inline simdjson_result<double> document::get_double() noexcept {
   return get_root_value_iterator().get_root_double(true);
@@ -152961,9 +196116,36 @@ simdjson_inline simdjson_result<double> document::get_double() noexcept {
 simdjson_inline simdjson_result<double> document::get_double_in_string() noexcept {
   return get_root_value_iterator().get_root_double_in_string(true);
 }
+simdjson_inline simdjson_result<float> document::get_float() noexcept {
+  return get_root_value_iterator().get_root_float(true);
+}
+simdjson_inline simdjson_result<float> document::get_float_in_string() noexcept {
+  return get_root_value_iterator().get_root_float_in_string(true);
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> document::get_float32() noexcept {
+  float result;
+  SIMDJSON_TRY(get_float().get(result));
+  return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> document::get_float64() noexcept {
+  double result;
+  SIMDJSON_TRY(get_double().get(result));
+  return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<std::string_view> document::get_string(bool allow_replacement) noexcept {
   return get_root_value_iterator().get_root_string(true, allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> document::get_u8string(bool allow_replacement) noexcept {
+  std::string_view content;
+  SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+  return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code document::get_string(string_type& receiver, bool allow_replacement) noexcept {
   return get_root_value_iterator().get_root_string(receiver, true, allow_replacement);
@@ -152985,11 +196167,25 @@ template<> simdjson_inline simdjson_result<array> document::get() & noexcept { r
 template<> simdjson_inline simdjson_result<object> document::get() & noexcept { return get_object(); }
 template<> simdjson_inline simdjson_result<raw_json_string> document::get() & noexcept { return get_raw_json_string(); }
 template<> simdjson_inline simdjson_result<std::string_view> document::get() & noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> document::get() & noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_inline simdjson_result<double> document::get() & noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> document::get() & noexcept { return get_float(); }
 template<> simdjson_inline simdjson_result<uint64_t> document::get() & noexcept { return get_uint64(); }
 template<> simdjson_inline simdjson_result<int64_t> document::get() & noexcept { return get_int64(); }
 template<> simdjson_inline simdjson_result<uint32_t> document::get() & noexcept { return get_uint32(); }
 template<> simdjson_inline simdjson_result<int32_t> document::get() & noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> document::get() & noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> document::get() & noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> document::get() & noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> document::get() & noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> document::get() & noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> document::get() & noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<> simdjson_inline simdjson_result<bool> document::get() & noexcept { return get_bool(); }
 template<> simdjson_inline simdjson_result<value> document::get() & noexcept { return get_value(); }

@@ -152997,17 +196193,35 @@ template<> simdjson_warn_unused simdjson_inline error_code document::get(array&
 template<> simdjson_warn_unused simdjson_inline error_code document::get(object& out) & noexcept { return get_object().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(raw_json_string& out) & noexcept { return get_raw_json_string().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(std::string_view& out) & noexcept { return get_string(false).get(out); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::u8string_view& out) & noexcept { return get_u8string(false).get(out); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_warn_unused simdjson_inline error_code document::get(double& out) & noexcept { return get_double().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(float& out) & noexcept { return get_float().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(uint64_t& out) & noexcept { return get_uint64().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(int64_t& out) & noexcept { return get_int64().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(uint32_t& out) & noexcept { return get_uint32().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(int32_t& out) & noexcept { return get_int32().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(uint16_t& out) & noexcept { return get_uint16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(int16_t& out) & noexcept { return get_int16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(uint8_t& out) & noexcept { return get_uint8().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(int8_t& out) & noexcept { return get_int8().get(out); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::float32_t& out) & noexcept { return get_float32().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::float64_t& out) & noexcept { return get_float64().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<> simdjson_warn_unused simdjson_inline error_code document::get(bool& out) & noexcept { return get_bool().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(value& out) & noexcept { return get_value().get(out); }

 template<> simdjson_deprecated simdjson_inline simdjson_result<raw_json_string> document::get() && noexcept { return get_raw_json_string(); }
 template<> simdjson_deprecated simdjson_inline simdjson_result<std::string_view> document::get() && noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_deprecated simdjson_inline simdjson_result<std::u8string_view> document::get() && noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_deprecated simdjson_inline simdjson_result<double> document::get() && noexcept { return std::forward<document>(*this).get_double(); }
+template<> simdjson_deprecated simdjson_inline simdjson_result<float> document::get() && noexcept { return std::forward<document>(*this).get_float(); }
 template<> simdjson_deprecated simdjson_inline simdjson_result<uint64_t> document::get() && noexcept { return std::forward<document>(*this).get_uint64(); }
 template<> simdjson_deprecated simdjson_inline simdjson_result<int64_t> document::get() && noexcept { return std::forward<document>(*this).get_int64(); }
 template<> simdjson_deprecated simdjson_inline simdjson_result<bool> document::get() && noexcept { return std::forward<document>(*this).get_bool(); }
@@ -153346,6 +196560,22 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<lsx::ondemand::document
   if (error()) { return error(); }
   return first.get_int32();
 }
+simdjson_inline simdjson_result<uint16_t> simdjson_result<lsx::ondemand::document>::get_uint16() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<lsx::ondemand::document>::get_int16() noexcept {
+  if (error()) { return error(); }
+  return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<lsx::ondemand::document>::get_uint8() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<lsx::ondemand::document>::get_int8() noexcept {
+  if (error()) { return error(); }
+  return first.get_int8();
+}
 simdjson_inline simdjson_result<double> simdjson_result<lsx::ondemand::document>::get_double() noexcept {
   if (error()) { return error(); }
   return first.get_double();
@@ -153354,10 +196584,36 @@ simdjson_inline simdjson_result<double> simdjson_result<lsx::ondemand::document>
   if (error()) { return error(); }
   return first.get_double_in_string();
 }
+simdjson_inline simdjson_result<float> simdjson_result<lsx::ondemand::document>::get_float() noexcept {
+  if (error()) { return error(); }
+  return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<lsx::ondemand::document>::get_float_in_string() noexcept {
+  if (error()) { return error(); }
+  return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<lsx::ondemand::document>::get_float32() noexcept {
+  if (error()) { return error(); }
+  return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<lsx::ondemand::document>::get_float64() noexcept {
+  if (error()) { return error(); }
+  return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<std::string_view> simdjson_result<lsx::ondemand::document>::get_string(bool allow_replacement) noexcept {
   if (error()) { return error(); }
   return first.get_string(allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<lsx::ondemand::document>::get_u8string(bool allow_replacement) noexcept {
+  if (error()) { return error(); }
+  return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code simdjson_result<lsx::ondemand::document>::get_string(string_type& receiver, bool allow_replacement) noexcept {
   if (error()) { return error(); }
@@ -153385,22 +196641,46 @@ simdjson_inline simdjson_result<bool> simdjson_result<lsx::ondemand::document>::
 }

 template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<lsx::ondemand::document>::get() & noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<lsx::ondemand::document>::get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lsx::ondemand::document>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>();
 }
 template<typename T>
-simdjson_deprecated simdjson_inline simdjson_result<T> simdjson_result<lsx::ondemand::document>::get() && noexcept {
+simdjson_deprecated simdjson_inline simdjson_result<T> simdjson_result<lsx::ondemand::document>::get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lsx::ondemand::document>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return std::forward<lsx::ondemand::document>(first).get<T>();
 }
 template<typename T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<lsx::ondemand::document>::get(T &out) & noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<lsx::ondemand::document>::get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lsx::ondemand::document>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>(out);
 }
 template<typename T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<lsx::ondemand::document>::get(T &out) && noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<lsx::ondemand::document>::get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lsx::ondemand::document>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return std::forward<lsx::ondemand::document>(first).get<T>(out);
 }
@@ -153469,27 +196749,27 @@ simdjson_inline simdjson_result<lsx::ondemand::document>::operator lsx::ondemand
 }
 simdjson_inline simdjson_result<lsx::ondemand::document>::operator uint64_t() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_uint64();
 }
 simdjson_inline simdjson_result<lsx::ondemand::document>::operator int64_t() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_int64();
 }
 simdjson_inline simdjson_result<lsx::ondemand::document>::operator double() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_double();
 }
 simdjson_inline simdjson_result<lsx::ondemand::document>::operator std::string_view() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_string();
 }
 simdjson_inline simdjson_result<lsx::ondemand::document>::operator lsx::ondemand::raw_json_string() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_raw_json_string();
 }
 simdjson_inline simdjson_result<lsx::ondemand::document>::operator bool() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_bool();
 }
 simdjson_inline simdjson_result<lsx::ondemand::document>::operator lsx::ondemand::value() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
@@ -153579,21 +196859,38 @@ simdjson_inline simdjson_result<uint64_t> document_reference::get_uint64() noexc
 simdjson_inline simdjson_result<uint64_t> document_reference::get_uint64_in_string() noexcept { return doc->get_root_value_iterator().get_root_uint64_in_string(false); }
 simdjson_inline simdjson_result<int64_t> document_reference::get_int64() noexcept { return doc->get_root_value_iterator().get_root_int64(false); }
 simdjson_inline simdjson_result<int64_t> document_reference::get_int64_in_string() noexcept { return doc->get_root_value_iterator().get_root_int64_in_string(false); }
-simdjson_inline simdjson_result<uint32_t> document_reference::get_uint32() noexcept {
-  uint64_t result;
-  SIMDJSON_TRY(doc->get_root_value_iterator().get_root_uint64(false).get(result));
-  if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<uint32_t>(result);
-}
-simdjson_inline simdjson_result<int32_t> document_reference::get_int32() noexcept {
-  int64_t result;
-  SIMDJSON_TRY(doc->get_root_value_iterator().get_root_int64(false).get(result));
-  if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<int32_t>(result);
-}
+simdjson_inline simdjson_result<uint32_t> document_reference::get_uint32() noexcept { return narrow_integer<uint32_t>(get_uint64()); }
+simdjson_inline simdjson_result<int32_t> document_reference::get_int32() noexcept { return narrow_integer<int32_t>(get_int64()); }
+simdjson_inline simdjson_result<uint16_t> document_reference::get_uint16() noexcept { return narrow_integer<uint16_t>(get_uint64()); }
+simdjson_inline simdjson_result<int16_t> document_reference::get_int16() noexcept { return narrow_integer<int16_t>(get_int64()); }
+simdjson_inline simdjson_result<uint8_t> document_reference::get_uint8() noexcept { return narrow_integer<uint8_t>(get_uint64()); }
+simdjson_inline simdjson_result<int8_t> document_reference::get_int8() noexcept { return narrow_integer<int8_t>(get_int64()); }
 simdjson_inline simdjson_result<double> document_reference::get_double() noexcept { return doc->get_root_value_iterator().get_root_double(false); }
-simdjson_inline simdjson_result<double> document_reference::get_double_in_string() noexcept { return doc->get_root_value_iterator().get_root_double(false); }
+simdjson_inline simdjson_result<double> document_reference::get_double_in_string() noexcept { return doc->get_root_value_iterator().get_root_double_in_string(false); }
+simdjson_inline simdjson_result<float> document_reference::get_float() noexcept { return doc->get_root_value_iterator().get_root_float(false); }
+simdjson_inline simdjson_result<float> document_reference::get_float_in_string() noexcept { return doc->get_root_value_iterator().get_root_float_in_string(false); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> document_reference::get_float32() noexcept {
+  float result;
+  SIMDJSON_TRY(get_float().get(result));
+  return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> document_reference::get_float64() noexcept {
+  double result;
+  SIMDJSON_TRY(get_double().get(result));
+  return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<std::string_view> document_reference::get_string(bool allow_replacement) noexcept { return doc->get_root_value_iterator().get_root_string(false, allow_replacement); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> document_reference::get_u8string(bool allow_replacement) noexcept {
+  std::string_view content;
+  SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+  return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code document_reference::get_string(string_type& receiver, bool allow_replacement) noexcept { return doc->get_root_value_iterator().get_root_string(receiver, false, allow_replacement); }
 simdjson_inline simdjson_result<std::string_view> document_reference::get_wobbly_string() noexcept { return doc->get_root_value_iterator().get_root_wobbly_string(false); }
@@ -153605,11 +196902,25 @@ template<> simdjson_inline simdjson_result<array> document_reference::get() & no
 template<> simdjson_inline simdjson_result<object> document_reference::get() & noexcept { return get_object(); }
 template<> simdjson_inline simdjson_result<raw_json_string> document_reference::get() & noexcept { return get_raw_json_string(); }
 template<> simdjson_inline simdjson_result<std::string_view> document_reference::get() & noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> document_reference::get() & noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_inline simdjson_result<double> document_reference::get() & noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> document_reference::get() & noexcept { return get_float(); }
 template<> simdjson_inline simdjson_result<uint64_t> document_reference::get() & noexcept { return get_uint64(); }
 template<> simdjson_inline simdjson_result<int64_t> document_reference::get() & noexcept { return get_int64(); }
 template<> simdjson_inline simdjson_result<uint32_t> document_reference::get() & noexcept { return get_uint32(); }
 template<> simdjson_inline simdjson_result<int32_t> document_reference::get() & noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> document_reference::get() & noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> document_reference::get() & noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> document_reference::get() & noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> document_reference::get() & noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> document_reference::get() & noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> document_reference::get() & noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<> simdjson_inline simdjson_result<bool> document_reference::get() & noexcept { return get_bool(); }
 template<> simdjson_inline simdjson_result<value> document_reference::get() & noexcept { return get_value(); }
 #if SIMDJSON_EXCEPTIONS
@@ -153755,6 +197066,22 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<lsx::ondemand::document
   if (error()) { return error(); }
   return first.get_int32();
 }
+simdjson_inline simdjson_result<uint16_t> simdjson_result<lsx::ondemand::document_reference>::get_uint16() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<lsx::ondemand::document_reference>::get_int16() noexcept {
+  if (error()) { return error(); }
+  return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<lsx::ondemand::document_reference>::get_uint8() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<lsx::ondemand::document_reference>::get_int8() noexcept {
+  if (error()) { return error(); }
+  return first.get_int8();
+}
 simdjson_inline simdjson_result<double> simdjson_result<lsx::ondemand::document_reference>::get_double() noexcept {
   if (error()) { return error(); }
   return first.get_double();
@@ -153763,10 +197090,36 @@ simdjson_inline simdjson_result<double> simdjson_result<lsx::ondemand::document_
   if (error()) { return error(); }
   return first.get_double_in_string();
 }
+simdjson_inline simdjson_result<float> simdjson_result<lsx::ondemand::document_reference>::get_float() noexcept {
+  if (error()) { return error(); }
+  return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<lsx::ondemand::document_reference>::get_float_in_string() noexcept {
+  if (error()) { return error(); }
+  return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<lsx::ondemand::document_reference>::get_float32() noexcept {
+  if (error()) { return error(); }
+  return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<lsx::ondemand::document_reference>::get_float64() noexcept {
+  if (error()) { return error(); }
+  return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<std::string_view> simdjson_result<lsx::ondemand::document_reference>::get_string(bool allow_replacement) noexcept {
   if (error()) { return error(); }
   return first.get_string(allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<lsx::ondemand::document_reference>::get_u8string(bool allow_replacement) noexcept {
+  if (error()) { return error(); }
+  return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code simdjson_result<lsx::ondemand::document_reference>::get_string(string_type& receiver, bool allow_replacement) noexcept {
   if (error()) { return error(); }
@@ -153793,22 +197146,46 @@ simdjson_inline simdjson_result<bool> simdjson_result<lsx::ondemand::document_re
   return first.is_null();
 }
 template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<lsx::ondemand::document_reference>::get() & noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<lsx::ondemand::document_reference>::get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lsx::ondemand::document_reference>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>();
 }
 template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<lsx::ondemand::document_reference>::get() && noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<lsx::ondemand::document_reference>::get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lsx::ondemand::document_reference>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return std::forward<lsx::ondemand::document_reference>(first).get<T>();
 }
 template <class T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<lsx::ondemand::document_reference>::get(T &out) & noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<lsx::ondemand::document_reference>::get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lsx::ondemand::document_reference>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>(out);
 }
 template <class T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<lsx::ondemand::document_reference>::get(T &out) && noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<lsx::ondemand::document_reference>::get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lsx::ondemand::document_reference>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return std::forward<lsx::ondemand::document_reference>(first).get<T>(out);
 }
@@ -153870,27 +197247,27 @@ simdjson_inline simdjson_result<lsx::ondemand::document_reference>::operator lsx
 }
 simdjson_inline simdjson_result<lsx::ondemand::document_reference>::operator uint64_t() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_uint64();
 }
 simdjson_inline simdjson_result<lsx::ondemand::document_reference>::operator int64_t() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_int64();
 }
 simdjson_inline simdjson_result<lsx::ondemand::document_reference>::operator double() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_double();
 }
 simdjson_inline simdjson_result<lsx::ondemand::document_reference>::operator std::string_view() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_string();
 }
 simdjson_inline simdjson_result<lsx::ondemand::document_reference>::operator lsx::ondemand::raw_json_string() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_raw_json_string();
 }
 simdjson_inline simdjson_result<lsx::ondemand::document_reference>::operator bool() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_bool();
 }
 simdjson_inline simdjson_result<lsx::ondemand::document_reference>::operator lsx::ondemand::value() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
@@ -153956,6 +197333,7 @@ simdjson_warn_unused simdjson_inline error_code simdjson_result<lsx::ondemand::d
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */

 #include <algorithm>
+#include <cstring>
 #include <stdexcept>

 namespace simdjson {
@@ -154042,23 +197420,20 @@ simdjson_inline document_stream::document_stream(
   const uint8_t *_buf,
   size_t _len,
   size_t _batch_size,
-  bool _allow_comma_separated
+  bool _allow_comma_separated,
+  stream_format _format
 ) noexcept
   : parser{&_parser},
     buf{_buf},
     len{_len},
     batch_size{_batch_size <= MINIMAL_BATCH_SIZE ? MINIMAL_BATCH_SIZE : _batch_size},
     allow_comma_separated{_allow_comma_separated},
+    format{_format},
     error{SUCCESS}
     #ifdef SIMDJSON_THREADS_ENABLED
     , use_thread(_parser.threaded) // we need to make a copy because _parser.threaded can change
     #endif
 {
-#ifdef SIMDJSON_THREADS_ENABLED
-  if(worker.get() == nullptr) {
-    error = MEMALLOC;
-  }
-#endif
 }

 simdjson_inline document_stream::document_stream() noexcept
@@ -154067,6 +197442,7 @@ simdjson_inline document_stream::document_stream() noexcept
     len{0},
     batch_size{0},
     allow_comma_separated{false},
+    format{stream_format::whitespace_delimited},
     error{UNINITIALIZED}
     #ifdef SIMDJSON_THREADS_ENABLED
     , use_thread(false)
@@ -154086,6 +197462,9 @@ inline size_t document_stream::size_in_bytes() const noexcept {
 }

 inline size_t document_stream::truncated_bytes() const noexcept {
+  // Stage 1 returns EMPTY on zero-length input before it writes the index
+  // sentinels read below, so they would still hold a previous stream's values.
+  if (len == 0) { return 0; }
   if(error == CAPACITY) { return len - batch_start; }
   return parser->implementation->structural_indexes[parser->implementation->n_structural_indexes] - parser->implementation->structural_indexes[parser->implementation->n_structural_indexes + 1];
 }
@@ -154166,13 +197545,20 @@ inline void document_stream::start() noexcept {
     error = run_stage1(*parser, batch_start);
   }
   if (error) { return; }
-  doc_index = batch_start;
+  // For json_sequence mode, structural_indexes[0] points to the actual JSON value
+  // after the RS delimiter and any following whitespace. For regular mode, it is
+  // the offset from batch_start to the first document in the batch.
+  doc_index = batch_start + parser->implementation->structural_indexes[0];
   doc = document(json_iterator(&buf[batch_start], parser));
   doc.iter._streaming = true;

   #ifdef SIMDJSON_THREADS_ENABLED
   if (use_thread && next_batch_start() < len) {
     // Kick off the first thread on next batch if needed
+    if (worker.get() == nullptr) {
+      worker.reset(new(std::nothrow) stage1_worker());
+      if (worker.get() == nullptr) { error = MEMALLOC; return; }
+    }
     error = stage1_thread_parser.allocate(batch_size);
     if (error) { return; }
     worker->start_thread();
@@ -154247,12 +197633,69 @@ inline void document_stream::next() noexcept {
        */

       if (error) { continue; } // If the error was EMPTY, we may want to load another batch.
-      doc_index = batch_start;
+      doc_index = batch_start + parser->implementation->structural_indexes[0];
     }
   }
 }

+simdjson_inline uint8_t document_stream::document_delimiter() const noexcept {
+  switch (format) {
+    case stream_format::newline_delimited: return '\n';
+    case stream_format::json_sequence: return 0x1E;
+    default: return 0;
+  }
+}
+
+simdjson_inline bool document_stream::skip_to_delimiter(uint8_t delimiter) noexcept {
+  const uint8_t *const base = &buf[batch_start];
+  const token_position pos = doc.iter.position();
+  const token_position end = doc.iter.end_position();
+  if (pos >= end) { return false; }
+  const size_t here = size_t(doc.iter.token.peek(pos) - base);
+  const size_t batch_len =
+      (len - batch_start < batch_size) ? len - batch_start : batch_size;
+  if (here >= batch_len) { return false; }
+  const uint8_t *const found = static_cast<const uint8_t *>(
+      std::memchr(base + here, delimiter, batch_len - here));
+  if (found == nullptr) { return false; }
+
+  const uint32_t boundary = uint32_t(found - base);
+  // The answer is near `pos`: the delimiter ends the current document, while
+  // `end` spans the whole batch. Gallop first so the cost follows the distance
+  // rather than the size of the batch.
+  token_position lo = pos;
+  size_t hop = 1;
+  while (lo + hop < end && lo[hop] < boundary) { lo += hop; hop <<= 1; }
+  token_position hi = (lo + hop < end) ? lo + hop : end;
+  while (lo < hi) {
+    const token_position mid = lo + ((hi - lo) >> 1);
+    if (*mid < boundary) { lo = mid + 1; } else { hi = mid; }
+  }
+  doc.iter.token.set_position(lo);
+  return true;
+}
+
 inline void document_stream::next_document() noexcept {
+  // A delimiter that cannot occur inside a document tells us where the current
+  // one ends, so we can jump there instead of walking every structural. Only
+  // valid while the iterator is still inside the document: a consumed document
+  // already sits on the next one's first token, and skip_child() returns at
+  // once for it.
+  //
+  // The jump does not structure-validate the unread remainder of the current
+  // document: under newline_delimited / json_sequence the next delimiter is
+  // assumed to be the true document boundary. Callers that leave depth() > 0
+  // while violating that contract (e.g. pretty multi-line JSON under
+  // newline_delimited) can mis-align following documents; use
+  // whitespace_delimited if unsure.
+  const uint8_t delimiter = document_delimiter();
+  if (delimiter != 0 && !error && doc.iter.depth() > 0 &&
+      skip_to_delimiter(delimiter)) {
+    doc.iter._depth = 1;
+    doc.iter._string_buf_loc = parser->string_buf.get();
+    doc.iter._root = doc.iter.position();
+    return;
+  }
   // Go to next place where depth=0 (document depth)
   error = doc.iter.skip_child(0);
   if (error) { return; }
@@ -154276,10 +197719,35 @@ inline error_code document_stream::run_stage1(ondemand::parser &p, size_t _batch
   // This code only updates the structural index in the parser, it does not update any json_iterator
   // instance.
   size_t remaining = len - _batch_start;
+  stage1_mode mode;
   if (remaining <= batch_size) {
-    return p.implementation->stage1(&buf[_batch_start], remaining, stage1_mode::streaming_final);
+    // Final batch
+    switch (format) {
+      case stream_format::json_sequence:
+        mode = stage1_mode::json_sequence_final;
+        break;
+      case stream_format::comma_delimited:
+        mode = stage1_mode::comma_delimited_final;
+        break;
+      default:
+        mode = stage1_mode::streaming_final;
+        break;
+    }
+    return p.implementation->stage1(&buf[_batch_start], remaining, mode);
   } else {
-    return p.implementation->stage1(&buf[_batch_start], batch_size, stage1_mode::streaming_partial);
+    // Partial batch
+    switch (format) {
+      case stream_format::json_sequence:
+        mode = stage1_mode::json_sequence_partial;
+        break;
+      case stream_format::comma_delimited:
+        mode = stage1_mode::comma_delimited_partial;
+        break;
+      default:
+        mode = stage1_mode::streaming_partial;
+        break;
+    }
+    return p.implementation->stage1(&buf[_batch_start], batch_size, mode);
   }
 }

@@ -154288,11 +197756,19 @@ simdjson_inline size_t document_stream::iterator::current_index() const noexcept
 }

 simdjson_inline std::string_view document_stream::iterator::source() const noexcept {
-  auto depth = stream->doc.iter.depth();
+  // On error (e.g., CAPACITY), there is no document to walk: return the rest of
+  // the input, as the DOM document_stream does.
+  if (stream->error) {
+    return std::string_view(reinterpret_cast<const char*>(stream->buf) + current_index(), stream->len - current_index());
+  }
+  // Always walk from the root of the document, whatever the current position
+  // of the document iterator: the user may have already consumed part of the
+  // document, so the iterator's current depth must not be used here.
+  depth_t depth = 1;
   auto cur_struct_index = stream->doc.iter._root - stream->parser->implementation->structural_indexes.get();

-  // If at root, process the first token to determine if scalar value
-  if (stream->doc.iter.at_root()) {
+  // Process the first token to determine if scalar value
+  {
     switch (stream->buf[stream->batch_start + stream->parser->implementation->structural_indexes[cur_struct_index]]) {
       case '{': case '[':   // Depth=1 already at start of document
         break;
@@ -154300,14 +197776,47 @@ simdjson_inline std::string_view document_stream::iterator::source() const noexc
         depth--;
         break;
       default:    // Scalar value document
-        // TODO: We could remove trailing whitespaces
         // This returns a string spanning from start of value to the beginning of the next document (excluded)
         {
           auto next_index = stream->batch_start + stream->parser->implementation->structural_indexes[++cur_struct_index];
           // normally the length would be next_index - current_index() - 1, except for the last document
           size_t svlen = next_index - current_index();
           const char *start = reinterpret_cast<const char*>(stream->buf) + current_index();
-          while(svlen > 1 && (std::isspace(start[svlen-1]) || start[svlen-1] == '\0')) {
+          // When the scalar is followed by a truncated document, the structural
+          // indexes of that document were dropped and next_index is the end of
+          // the input, so we bound the scalar by scanning the token itself.
+          size_t token_len = 0;
+          if (*start == '"') {
+            token_len = 1;
+            while (token_len < svlen) {
+              char c = start[token_len++];
+              if (c == '\\') {
+                token_len++;
+              } else if (c == '"') {
+                break;
+              }
+            }
+          } else {
+            while (token_len < svlen) {
+              char c = start[token_len];
+              if (std::isspace(static_cast<unsigned char>(c)) || c == ',' || c == '{' || c == '[' || c == '\0' || static_cast<uint8_t>(c) == 0x1E) {
+                break;
+              }
+              token_len++;
+            }
+          }
+          if (token_len > 0 && token_len < svlen) {
+            svlen = token_len;
+          }
+          // Trim trailing whitespace, NUL, and RS (0x1E). In RFC 7464
+          // json_sequence mode the scanner classifies RS as a scalar
+          // character, so an RS-prefixed scalar document (number / true /
+          // false / null / string) has no closing structural index and the
+          // slice runs all the way up to the next document's RS. RS cannot
+          // legally appear in a JSON value at the source level (control
+          // characters in strings must be escaped as \u001E), so stripping
+          // it is safe in every stream_format.
+          while(svlen > 1 && (std::isspace(static_cast<unsigned char>(start[svlen-1])) || start[svlen-1] == '\0' || static_cast<uint8_t>(start[svlen-1]) == 0x1E || (stream->format == stream_format::comma_delimited && start[svlen-1] == ','))) {
             svlen--;
           }
           return std::string_view(start, svlen);
@@ -154432,11 +197941,19 @@ simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> field::un
   return answer;
 }

+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_warn_unused simdjson_result<std::u8string_view> field::unescaped_u8key(bool allow_replacement) noexcept {
+  std::string_view key;
+  SIMDJSON_TRY( unescaped_key(allow_replacement).get(key) );
+  return internal::as_u8string_view(key);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 template <typename string_type>
 simdjson_inline simdjson_warn_unused error_code field::unescaped_key(string_type& receiver, bool allow_replacement) noexcept {
   std::string_view key;
   SIMDJSON_TRY( unescaped_key(allow_replacement).get(key) );
-  receiver = key;
+  internal::assign_utf8(receiver, key);
   return SUCCESS;
 }

@@ -154458,6 +197975,12 @@ simdjson_inline std::string_view field::escaped_key() const noexcept {
   return std::string_view(reinterpret_cast<const char*>(first.buf), end_quote - first.buf);
 }

+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline std::u8string_view field::escaped_u8key() const noexcept {
+  return internal::as_u8string_view(escaped_key());
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 simdjson_inline value &field::value() & noexcept {
   return second;
 }
@@ -154502,11 +198025,25 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<lsx::ondemand:
   return first.escaped_key();
 }

+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<lsx::ondemand::field>::escaped_u8key() noexcept {
+  if (error()) { return error(); }
+  return first.escaped_u8key();
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 simdjson_inline simdjson_result<std::string_view> simdjson_result<lsx::ondemand::field>::unescaped_key(bool allow_replacement) noexcept {
   if (error()) { return error(); }
   return first.unescaped_key(allow_replacement);
 }

+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<lsx::ondemand::field>::unescaped_u8key(bool allow_replacement) noexcept {
+  if (error()) { return error(); }
+  return first.unescaped_u8key(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 template<typename string_type>
 simdjson_warn_unused simdjson_inline error_code simdjson_result<lsx::ondemand::field>::unescaped_key(string_type &receiver, bool allow_replacement) noexcept {
   if (error()) { return error(); }
@@ -154550,6 +198087,9 @@ simdjson_inline json_iterator::json_iterator(json_iterator &&other) noexcept
     _depth{other._depth},
     _root{other._root},
     _streaming{other._streaming}
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+    , _allow_incomplete_json{other._allow_incomplete_json}
+#endif
 {
   other.parser = nullptr;
 }
@@ -154561,6 +198101,9 @@ simdjson_inline json_iterator &json_iterator::operator=(json_iterator &&other) n
   _depth = other._depth;
   _root = other._root;
   _streaming = other._streaming;
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  _allow_incomplete_json = other._allow_incomplete_json;
+#endif
   other.parser = nullptr;
   return *this;
 }
@@ -154587,7 +198130,8 @@ simdjson_inline json_iterator::json_iterator(const uint8_t *buf, ondemand::parse
       _string_buf_loc{parser->string_buf.get()},
       _depth{1},
       _root{parser->implementation->structural_indexes.get()},
-      _streaming{streaming}
+      _streaming{streaming},
+      _allow_incomplete_json{true}

 {
   logger::log_headers();
@@ -154659,7 +198203,8 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::skip_child(depth_
 #endif // SIMDJSON_CHECK_EOF
       break;
     case '"':
-      if(*peek() == ':') {
+      // At the end, peek() would read the sentinel, which points into the padding.
+      if(!at_end() && *peek() == ':') {
         // We are at a key!!!
         // This might happen if you just started an object and you skip it immediately.
         // Performance note: it would be nice to get rid of this check as it is somewhat
@@ -154702,7 +198247,7 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::skip_child(depth_
     }
   }

-  return report_error(TAPE_ERROR, "not enough close braces");
+  return report_error(INCOMPLETE_ARRAY_OR_OBJECT, "not enough close braces");
 }

 SIMDJSON_POP_DISABLE_WARNINGS
@@ -154719,6 +198264,17 @@ simdjson_inline bool json_iterator::streaming() const noexcept {
   return _streaming;
 }

+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+simdjson_inline bool json_iterator::allow_incomplete_json() const noexcept {
+  return _allow_incomplete_json;
+}
+
+simdjson_inline size_t json_iterator::remaining_input_length(const uint8_t *json) const noexcept {
+  const uint8_t *end = token.buf + parser->_document_len;
+  return json < end ? size_t(end - json) : 0;
+}
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+
 simdjson_inline token_position json_iterator::root_position() const noexcept {
   return _root;
 }
@@ -155001,7 +198557,7 @@ inline std::ostream& operator<<(std::ostream& out, json_type type) noexcept {
         case json_type::string: out << "string"; break;
         case json_type::boolean: out << "boolean"; break;
         case json_type::null: out << "null"; break;
-        default: SIMDJSON_UNREACHABLE();
+        case json_type::unknown: out << "unknown"; break;
     }
     return out;
 }
@@ -155340,6 +198896,10 @@ inline void log_line(const json_iterator &iter, token_position index, depth_t de
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_iterator.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value-inl.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/jsonpathutil.h" */
+/* amalgamation skipped (editor-only): #include <utility> */
+/* amalgamation skipped (editor-only): #if SIMDJSON_SUPPORTS_CONCEPTS */
+/* amalgamation skipped (editor-only): #include <tuple> // std::forward_as_tuple/get for the variadic for_each adapter */
+/* amalgamation skipped (editor-only): #endif */
 /* amalgamation skipped (editor-only): #if SIMDJSON_STATIC_REFLECTION */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_string_builder.h"  // for constevalutil::fixed_string */
 /* amalgamation skipped (editor-only): #include <meta> */
@@ -155369,12 +198929,21 @@ simdjson_inline simdjson_result<value> object::find_field_unordered(const std::s
   return value(iter.child());
 }
 simdjson_inline simdjson_result<value> object::operator[](const std::string_view key) & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   return find_field_unordered(key);
 }
 simdjson_inline simdjson_result<value> object::operator[](const std::string_view key) && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   return std::forward<object>(*this).find_field_unordered(key);
 }
 simdjson_inline simdjson_result<value> object::find_field(const std::string_view key) & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   bool has_value;
   SIMDJSON_TRY( iter.find_field_raw(key).get(has_value) );
   if (!has_value) {
@@ -155384,6 +198953,9 @@ simdjson_inline simdjson_result<value> object::find_field(const std::string_view
   return value(iter.child());
 }
 simdjson_inline simdjson_result<value> object::find_field(const std::string_view key) && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   bool has_value;
   SIMDJSON_TRY( iter.find_field_raw(key).get(has_value) );
   if (!has_value) {
@@ -155393,6 +198965,150 @@ simdjson_inline simdjson_result<value> object::find_field(const std::string_view
   return value(iter.child());
 }

+#if SIMDJSON_SUPPORTS_CONCEPTS
+template <typename Selector, typename Func>
+  requires key_selector_type<Selector> &&
+           std::is_invocable_v<Func&, std::size_t, value>
+simdjson_flatten simdjson_inline for_each_result object::for_each(Func&& on_match)
+    noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, value>) {
+  // Single pass driven directly by the value_iterator, mirroring
+  // find_field_unordered_raw + value(iter.child()). Compared to walking via
+  // object_iterator/field, this avoids constructing a simdjson_result<field> and
+  // a field (key + value) for every field -- and the development-check bookkeeping
+  // in object_iterator -- building a value only for the fields that actually match.
+  // We operate on a copy of the iterator, as object::begin() would.
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  // Mirror object::begin(): for_each must start at the beginning of the object,
+  // not from some position left behind by a prior find_field on the same object.
+  if (!iter.is_at_iterator_start()) { return {OUT_OF_ORDER_ITERATION, 0}; }
+#endif
+  value_iterator it = iter;
+  std::size_t matched = 0;
+  // Track which selector indices have already matched, as a compile-time bitset
+  // (one 64-bit word per 64 keys): the callback fires once per key, on its first
+  // occurrence, and we stop as soon as every key has matched.
+  constexpr std::size_t seen_words = (Selector::size() + 63) / 64;
+  std::array<std::uint64_t, seen_words> seen{};
+  while (it.is_open()) {
+    raw_json_string key;
+    error_code error;
+    std::size_t idx;
+    if constexpr (Selector::window.ok) {
+      // A window selector confirms a key from its raw bytes alone (the closing
+      // quote bounds it), so we take the length-free path: field_key (no backward
+      // scan, mirroring the ordered find_field) feeding match_raw(raw_json_string).
+      if ((error = it.field_key().get(key))) { it.abandon(); return {error, matched}; }
+      if ((error = it.field_value())) { it.abandon(); return {error, matched}; }
+      idx = Selector::match_raw(key);
+    } else {
+      // Otherwise derive the key length from the structural index (the following
+      // ':' token) to feed the length-aware hash, avoiding a forward SIMD scan.
+      std::size_t key_len;
+      if ((error = it.field_key_with_length(key, key_len))) { it.abandon(); return {error, matched}; }
+      if ((error = it.field_value())) { it.abandon(); return {error, matched}; }
+      idx = Selector::match_raw(key.raw(), key_len);
+    }
+    if (idx < Selector::size()) {
+      const std::uint64_t seen_bit = std::uint64_t{1} << (idx & 63);
+      std::uint64_t &seen_word = seen[idx >> 6];
+      if (!(seen_word & seen_bit)) {
+        seen_word |= seen_bit;
+        value matched_value(it.child());
+        // The callback may return void or anything convertible to error_code
+        // (error_code itself, or a for_each_result from a nested for_each). When
+        // it yields an error_code, we stop at the first non-SUCCESS result and
+        // propagate it so the caller can surface value-parse errors (for example,
+        // a type mismatch on a matched field). A void-returning callback is
+        // responsible for handling its own errors.
+        if constexpr (std::is_convertible_v<decltype(on_match(idx, matched_value)), error_code>) {
+          // Unlike the internal-error paths above, a callback error does not
+          // abandon the iterator: we leave it recoverable so the caller can keep
+          // using the object (or its parent) after handling the error.
+          if ((error = on_match(idx, matched_value))) { return {error, matched}; }
+        } else {
+          on_match(idx, matched_value);
+        }
+        if (++matched >= Selector::size()) { break; }
+      }
+    }
+    // Mirror object_iterator::operator++'s safety rail: if the callback consumed
+    // the value and left the iterator closed or in error (e.g. a void callback
+    // that swallowed a fatal sub-iteration error), stop here rather than calling
+    // skip_child on a closed iterator.
+    if (!it.is_open()) { break; }
+    // Skip the value (a no-op if the callback consumed it) and step to the next
+    // field; has_next_field() ends the container on '}', which closes the loop.
+    if ((error = it.skip_child())) { it.abandon(); return {error, matched}; }
+    if ((error = it.has_next_field().error())) { return {error, matched}; }
+  }
+  return {SUCCESS, matched};
+}
+
+namespace key_selector_for_each_detail {
+
+// Dispatch a matched value to the I-th handler in the tuple (0-based). A handler
+// is either an invocable taking the value (void- or error_code-returning) or a
+// deserialization target, in which case we assign via value::get. We always
+// return error_code so the core (index, value) for_each can treat the adapter
+// uniformly.
+template <typename Tuple, std::size_t... Is>
+simdjson_really_inline error_code dispatch_value(
+    std::size_t idx, Tuple& handlers, value v, std::index_sequence<Is...>) {
+  error_code err = SUCCESS;
+  auto try_one = [&](auto Ic) {
+    constexpr std::size_t I = decltype(Ic)::value;
+    if (idx == I) {
+      auto&& h = std::get<I>(handlers);
+      using H = std::remove_reference_t<decltype(h)>;
+      if constexpr (std::is_invocable_v<H&, value>) {
+        // A handler returning void runs for its side effects; one returning
+        // anything convertible to error_code (error_code, or a for_each_result
+        // from a nested for_each) has its error captured and propagated.
+        if constexpr (std::is_convertible_v<decltype(h(v)), error_code>) {
+          err = h(v);
+        } else {
+          h(v);
+        }
+      } else {
+        // Direct deserialization target: assign the matched value into it.
+        err = v.get(h);
+      }
+    }
+  };
+  (try_one(std::integral_constant<std::size_t, Is>{}), ...);
+  return err;
+}
+
+} // namespace key_selector_for_each_detail
+
+template <typename Selector, typename... Handlers>
+  requires key_selector_type<Selector> &&
+           (sizeof...(Handlers) == Selector::size()) &&
+           (key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_flatten simdjson_inline for_each_result object::for_each(Handlers&&... on_match)
+    noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+  auto handlers = std::forward_as_tuple(std::forward<Handlers>(on_match)...);
+  // Reuse the single (index, value) implementation via a tiny adapter.
+  // The adapter is called once per *matched* key (very few); the hot path
+  // (iteration + match_raw + seen bitset) stays exactly the same.
+  return this->template for_each<Selector>(
+      [&](std::size_t i, value v) -> error_code {
+        return key_selector_for_each_detail::dispatch_value(
+            i, handlers, v, std::make_index_sequence<Selector::size()>{});
+      });
+}
+
+template <constevalutil::fixed_string... Keys, typename... Handlers>
+  requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+           (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+           (key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline for_each_result object::for_each(Handlers&&... on_match)
+    noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+  using Selector = key_selector<Keys...>;
+  return this->template for_each<Selector>(std::forward<Handlers>(on_match)...);
+}
+#endif
+
 simdjson_inline simdjson_result<object> object::start(value_iterator &iter) noexcept {
   SIMDJSON_TRY( iter.start_object().error() );
   return object(iter);
@@ -155428,6 +199144,9 @@ simdjson_warn_unused simdjson_inline error_code object::consume() noexcept {
 }

 simdjson_inline simdjson_result<std::string_view> object::raw_json() noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   const uint8_t * starting_point{iter.peek_start()};
   auto error = consume();
   if(error) { return error; }
@@ -155449,9 +199168,17 @@ simdjson_inline object::object(const value_iterator &_iter) noexcept
 {
 }

-simdjson_inline simdjson_result<object_iterator> object::begin() noexcept {
+simdjson_inline simdjson_result<object_iterator> object::begin() & noexcept {
 #if SIMDJSON_DEVELOPMENT_CHECKS
-  if (!iter.is_at_iterator_start()) { return OUT_OF_ORDER_ITERATION; }
+  if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
+  return object_iterator(iter, this);
+#endif
+  return object_iterator(iter);
+}
+simdjson_inline simdjson_result<object_iterator> object::begin() && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  // The object is a temporary that the iterator may outlive: do not lock it.
+  if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
 #endif
   return object_iterator(iter);
 }
@@ -155460,7 +199187,9 @@ simdjson_inline simdjson_result<object_iterator> object::end() noexcept {
 }

 inline simdjson_result<value> object::at_pointer(std::string_view json_pointer) noexcept {
-  if (json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
+  // An empty pointer has no json_pointer[0]: with a default-constructed
+  // std::string_view, reading it dereferences a null pointer.
+  if (json_pointer.empty() || json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
   json_pointer = json_pointer.substr(1);
   size_t slash = json_pointer.find('/');
   std::string_view key = json_pointer.substr(0, slash);
@@ -155562,6 +199291,9 @@ simdjson_inline simdjson_result<size_t> object::count_fields() & noexcept {
 }

 simdjson_inline simdjson_result<bool> object::is_empty() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   bool is_not_empty;
   auto error = iter.reset_object().get(is_not_empty);
   if(error) { return error; }
@@ -155569,9 +199301,41 @@ simdjson_inline simdjson_result<bool> object::is_empty() & noexcept {
 }

 simdjson_inline simdjson_result<bool> object::reset() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   return iter.reset_object();
 }

+simdjson_inline object_position object::get_current_position() const noexcept {
+  return object_position{iter.position(), iter.json_iter().depth()};
+}
+
+simdjson_inline error_code object::revert_position(object_position position) noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
+  // json_iterator::reenter_child() requires the live depth to be exactly
+  // one level shallower than the target (matching how every other depth
+  // transition in this iterator works), and under SIMDJSON_DEVELOPMENT_CHECKS
+  // additionally validates against the parser's per-depth container-start
+  // bookkeeping. Neither applies here: depending on what was captured and
+  // what has happened since (a scalar field fully consumed, a compound
+  // value left open, a find_field() miss that scanned past everything),
+  // the live depth when reverting can be any number of levels away from
+  // the captured one, and the captured depth is not necessarily a
+  // container's own start. reenter_at() moves directly, matching how
+  // reset_object() itself repositions without going through reenter_child().
+  iter.reenter_at(position.position, position.depth);
+  return SUCCESS;
+}
+
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline void object::set_locked(bool _locked) noexcept {
+  locked = _locked;
+}
+#endif
+
 #if SIMDJSON_SUPPORTS_CONCEPTS && SIMDJSON_STATIC_REFLECTION

 template<constevalutil::fixed_string... FieldNames, typename T>
@@ -155629,10 +199393,14 @@ simdjson_inline simdjson_result<lsx::ondemand::object>::simdjson_result(lsx::ond
 simdjson_inline simdjson_result<lsx::ondemand::object>::simdjson_result(error_code error) noexcept
     : implementation_simdjson_result_base<lsx::ondemand::object>(error) {}

-simdjson_inline simdjson_result<lsx::ondemand::object_iterator> simdjson_result<lsx::ondemand::object>::begin() noexcept {
+simdjson_inline simdjson_result<lsx::ondemand::object_iterator> simdjson_result<lsx::ondemand::object>::begin() & noexcept {
   if (error()) { return error(); }
   return first.begin();
 }
+simdjson_inline simdjson_result<lsx::ondemand::object_iterator> simdjson_result<lsx::ondemand::object>::begin() && noexcept {
+  if (error()) { return error(); }
+  return std::move(first).begin();
+}
 simdjson_inline simdjson_result<lsx::ondemand::object_iterator> simdjson_result<lsx::ondemand::object>::end() noexcept {
   if (error()) { return error(); }
   return first.end();
@@ -155686,11 +199454,55 @@ simdjson_inline error_code simdjson_result<lsx::ondemand::object>::for_each_at_p
   return first.for_each_at_path_with_wildcard(json_path, std::forward<Func>(callback));
 }

+#if SIMDJSON_SUPPORTS_CONCEPTS
+template <typename Selector, typename Func>
+  requires lsx::ondemand::key_selector_type<Selector> &&
+           std::is_invocable_v<Func&, std::size_t, lsx::ondemand::value>
+simdjson_inline lsx::ondemand::for_each_result
+simdjson_result<lsx::ondemand::object>::for_each(Func&& on_match)
+    noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, lsx::ondemand::value>) {
+  if (error()) { return {error(), 0}; }
+  return first.template for_each<Selector>(std::forward<Func>(on_match));
+}
+
+template <typename Selector, typename... Handlers>
+  requires lsx::ondemand::key_selector_type<Selector> &&
+           (sizeof...(Handlers) == Selector::size()) &&
+           (lsx::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline lsx::ondemand::for_each_result
+simdjson_result<lsx::ondemand::object>::for_each(Handlers&&... on_match)
+    noexcept(lsx::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+  if (error()) { return {error(), 0}; }
+  return first.template for_each<Selector>(std::forward<Handlers>(on_match)...);
+}
+
+template <constevalutil::fixed_string... Keys, typename... Handlers>
+  requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+           (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+           (lsx::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline lsx::ondemand::for_each_result
+simdjson_result<lsx::ondemand::object>::for_each(Handlers&&... on_match)
+    noexcept(lsx::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+  if (error()) { return {error(), 0}; }
+  return first.template for_each<Keys...>(std::forward<Handlers>(on_match)...);
+}
+#endif
+
 inline simdjson_result<bool> simdjson_result<lsx::ondemand::object>::reset() noexcept {
   if (error()) { return error(); }
   return first.reset();
 }

+inline simdjson_result<lsx::ondemand::object_position> simdjson_result<lsx::ondemand::object>::get_current_position() noexcept {
+  if (error()) { return error(); }
+  return first.get_current_position();
+}
+
+inline error_code simdjson_result<lsx::ondemand::object>::revert_position(lsx::ondemand::object_position position) noexcept {
+  if (error()) { return error(); }
+  return first.revert_position(position);
+}
+
 inline simdjson_result<bool> simdjson_result<lsx::ondemand::object>::is_empty() noexcept {
   if (error()) { return error(); }
   return first.is_empty();
@@ -155734,6 +199546,61 @@ simdjson_inline object_iterator::object_iterator(const value_iterator &_iter) no
   : iter{_iter}
 {}

+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline object_iterator::object_iterator(const value_iterator &_iter, object* _parent) noexcept
+  : parent{_parent}, iter{_iter}
+{
+  if (parent) parent->set_locked(true);
+}
+#endif
+
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline object_iterator::~object_iterator() noexcept
+{
+  if (parent) parent->set_locked(false);
+}
+
+simdjson_inline object_iterator::object_iterator(object_iterator&& other) noexcept
+  : has_been_referenced{other.has_been_referenced},
+    parent{other.parent},
+    iter{std::move(other.iter)}
+{
+  other.parent = nullptr;
+}
+
+simdjson_inline object_iterator& object_iterator::operator=(object_iterator&& other) noexcept {
+  if (this != &other)
+  {
+    if (parent)
+      parent->set_locked(false);
+    has_been_referenced = other.has_been_referenced;
+    parent = other.parent;
+    iter = std::move(other.iter);
+
+    other.parent = nullptr;
+  }
+  return *this;
+}
+
+simdjson_inline object_iterator::object_iterator(const object_iterator& other) noexcept
+  : has_been_referenced{other.has_been_referenced},
+    parent{nullptr},
+    iter{other.iter}
+{}
+
+simdjson_inline object_iterator& object_iterator::operator=(const object_iterator& other) noexcept {
+  if (this != &other)
+  {
+    if (parent)
+      parent->set_locked(false);
+    has_been_referenced = other.has_been_referenced;
+    parent = nullptr;
+    iter = other.iter;
+  }
+  return *this;
+}
+#endif
+
 simdjson_inline simdjson_result<field> object_iterator::operator*() noexcept {
 #if SIMDJSON_DEVELOPMENT_CHECKS
   // We must call * once per iteration.
@@ -155861,6 +199728,147 @@ simdjson_inline simdjson_result<lsx::ondemand::object_iterator> &simdjson_result

 #endif // SIMDJSON_GENERIC_ONDEMAND_OBJECT_ITERATOR_INL_H
 /* end file simdjson/generic/ondemand/object_iterator-inl.h for lsx */
+/* including simdjson/generic/ondemand/ranges-inl.h for lsx: #include "simdjson/generic/ondemand/ranges-inl.h" */
+/* begin file simdjson/generic/ondemand/ranges-inl.h for lsx */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/ranges.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array_iterator-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object_iterator-inl.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#if SIMDJSON_SUPPORTS_RANGES
+
+namespace simdjson {
+namespace lsx {
+namespace ondemand {
+
+//
+// array_range_iterator
+//
+
+simdjson_inline array_range_iterator::array_range_iterator(array_iterator iter) noexcept
+  : iter_{iter} {}
+
+simdjson_inline simdjson_result<value> array_range_iterator::operator*() const noexcept {
+  return *iter_;
+}
+
+simdjson_inline array_range_iterator& array_range_iterator::operator++() noexcept {
+  ++iter_;
+  return *this;
+}
+
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+simdjson_inline void array_range_iterator::operator++(int) noexcept {
+  ++*this;
+}
+SIMDJSON_POP_DISABLE_WARNINGS
+
+//
+// array_range
+//
+
+simdjson_inline array_range::array_range(array& arr) noexcept {
+  auto b = arr.begin();
+  if (b.error()) { error_ = b.error(); return; }
+  begin_ = b.value_unsafe();
+  end_ = arr.end().value_unsafe();
+}
+
+simdjson_inline array_range_iterator array_range::begin() noexcept {
+  return array_range_iterator(begin_);
+}
+
+simdjson_inline array_range_iterator array_range::end() noexcept {
+  return array_range_iterator(end_);
+}
+
+//
+// object_range_iterator
+//
+
+simdjson_inline object_range_iterator::object_range_iterator(object_iterator iter) noexcept
+  : iter_{iter} {}
+
+simdjson_inline simdjson_result<field> object_range_iterator::operator*() const noexcept {
+  return *iter_;
+}
+
+simdjson_inline object_range_iterator& object_range_iterator::operator++() noexcept {
+  ++iter_;
+  return *this;
+}
+
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+simdjson_inline void object_range_iterator::operator++(int) noexcept {
+  ++*this;
+}
+SIMDJSON_POP_DISABLE_WARNINGS
+
+
+//
+// object_range
+//
+
+simdjson_inline object_range::object_range(object& obj) noexcept {
+  auto b = obj.begin();
+  if (b.error()) { error_ = b.error(); return; }
+  begin_ = b.value_unsafe();
+  end_ = obj.end().value_unsafe();
+}
+
+simdjson_inline object_range_iterator object_range::begin() noexcept {
+  return object_range_iterator(begin_);
+}
+
+simdjson_inline object_range_iterator object_range::end() noexcept {
+  return object_range_iterator(end_);
+}
+
+//
+// Free functions
+//
+
+simdjson_inline array_range get_range(array& arr) noexcept {
+  return array_range(arr);
+}
+
+simdjson_inline object_range get_key_value_range(object& obj) noexcept {
+  return object_range(obj);
+}
+
+#if SIMDJSON_EXCEPTIONS
+simdjson_inline array_range get_range(simdjson_result<array> result) {
+  return array_range(result.value());
+}
+
+simdjson_inline object_range get_key_value_range(simdjson_result<object> result) {
+  return object_range(result.value());
+}
+#endif // SIMDJSON_EXCEPTIONS
+
+} // namespace ondemand
+} // namespace lsx
+} // namespace simdjson
+
+// Verify the range wrapper types satisfy the expected C++20 concepts.
+static_assert(std::input_iterator<simdjson::lsx::ondemand::array_range_iterator>);
+static_assert(std::input_iterator<simdjson::lsx::ondemand::object_range_iterator>);
+static_assert(std::ranges::input_range<simdjson::lsx::ondemand::array_range>);
+static_assert(std::ranges::input_range<simdjson::lsx::ondemand::object_range>);
+static_assert(std::ranges::view<simdjson::lsx::ondemand::array_range>);
+static_assert(std::ranges::view<simdjson::lsx::ondemand::object_range>);
+
+#endif // SIMDJSON_SUPPORTS_RANGES
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H
+/* end file simdjson/generic/ondemand/ranges-inl.h for lsx */
 /* including simdjson/generic/ondemand/parser-inl.h for lsx: #include "simdjson/generic/ondemand/parser-inl.h" */
 /* begin file simdjson/generic/ondemand/parser-inl.h for lsx */
 #ifndef SIMDJSON_GENERIC_ONDEMAND_PARSER_INL_H
@@ -155892,7 +199900,10 @@ simdjson_warn_unused simdjson_inline error_code parser::allocate(size_t new_capa

   // string_capacity copied from document::allocate
   _capacity = 0;
-  size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * new_capacity / 3 + SIMDJSON_PADDING, 64);
+  if(5 * (new_capacity / 3) + SIMDJSON_PADDING < SIMDJSON_PADDING) {
+    return CAPACITY; // overflow, only happen on legacy 32-bit systems with very large capacity
+  }
+  size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * (new_capacity / 3) + SIMDJSON_PADDING, 64);
   string_buf.reset(new (std::nothrow) uint8_t[string_capacity]);
 #if SIMDJSON_DEVELOPMENT_CHECKS
   start_positions.reset(new (std::nothrow) token_position[new_max_depth]);
@@ -155917,6 +199928,7 @@ simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate(p
   if (!json.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }

   json.remove_utf8_bom();
+  _document_len = json.length();

   // Allocate if needed
   if (capacity() < json.length() || !string_buf) {
@@ -155933,6 +199945,7 @@ simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate_a
   if (!json.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }

   json.remove_utf8_bom();
+  _document_len = json.length();

   // Allocate if needed
   if (capacity() < json.length() || !string_buf) {
@@ -155998,6 +200011,34 @@ simdjson_warn_unused simdjson_inline simdjson_result<json_iterator> parser::iter
   return json_iterator(reinterpret_cast<const uint8_t *>(json.data()), this);
 }

+inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size) noexcept {
+  if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
+  if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+    buf += 3;
+    len -= 3;
+  }
+  return document_stream(*this, buf, len, batch_size, false, stream_format::whitespace_delimited);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size) noexcept {
+  return iterate_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(padded_string_view s, size_t batch_size) noexcept {
+  if (!s.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
+  return iterate_many(s.data(), s.length(), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const padded_string &s, size_t batch_size) noexcept {
+  return iterate_many(padded_string_view(s), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const std::string &s, size_t batch_size) noexcept {
+  return iterate_many(padded_string_view(s), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size) noexcept {
+  return iterate_many(pad(s), batch_size);
+}
+
+#ifndef SIMDJSON_DISABLE_DEPRECATED_API
+SIMDJSON_PUSH_DISABLE_WARNINGS
+SIMDJSON_DISABLE_DEPRECATED_WARNING
 inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept {
   // Warning: no check is done on the buffer padding. We trust the user.
   if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
@@ -156005,8 +200046,11 @@ inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf,
     buf += 3;
     len -= 3;
   }
-  if(allow_comma_separated && batch_size < len) { batch_size = len; }
-  return document_stream(*this, buf, len, batch_size, allow_comma_separated);
+  // Map allow_comma_separated to stream_format::comma_delimited
+  if (allow_comma_separated) {
+    return document_stream(*this, buf, len, batch_size, false, stream_format::comma_delimited);
+  }
+  return document_stream(*this, buf, len, batch_size, false, stream_format::whitespace_delimited);
 }

 inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept {
@@ -156026,6 +200070,51 @@ inline simdjson_result<document_stream> parser::iterate_many(const std::string &
 inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size, bool allow_comma_separated) noexcept {
   return iterate_many(pad(s), batch_size, allow_comma_separated);
 }
+SIMDJSON_POP_DISABLE_WARNINGS
+#endif // SIMDJSON_DISABLE_DEPRECATED_API
+
+inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+  if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
+  if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+    buf += 3;
+    len -= 3;
+  }
+  if (format == stream_format::comma_delimited_array) {
+    // Strip leading JSON whitespace.
+    while (len > 0 && (buf[0] == ' ' || buf[0] == '\t' || buf[0] == '\n' || buf[0] == '\r')) {
+      buf++; len--;
+    }
+    // Expect the opening '['.
+    if (len == 0 || buf[0] != '[') { return TAPE_ERROR; }
+    buf++; len--;
+    // Strip trailing JSON whitespace.
+    while (len > 0 && (buf[len-1] == ' ' || buf[len-1] == '\t' || buf[len-1] == '\n' || buf[len-1] == '\r')) {
+      len--;
+    }
+    // Expect the closing ']'.
+    if (len == 0 || buf[len-1] != ']') { return TAPE_ERROR; }
+    len--;
+    // Fall through to comma_delimited over the array contents.
+    format = stream_format::comma_delimited;
+  }
+  return document_stream(*this, buf, len, batch_size, false, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+  return iterate_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(padded_string_view s, size_t batch_size, stream_format format) noexcept {
+  if (!s.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
+  return iterate_many(s.data(), s.length(), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const std::string &s, size_t batch_size, stream_format format) noexcept {
+  return iterate_many(padded_string_view(s), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size, stream_format format) noexcept {
+  return iterate_many(pad(s), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const padded_string &s, size_t batch_size, stream_format format) noexcept {
+  return iterate_many(padded_string_view(s), batch_size, format);
+}
 simdjson_pure simdjson_inline size_t parser::capacity() const noexcept {
   return _capacity;
 }
@@ -156433,6 +200522,27 @@ namespace simdjson {
 namespace lsx {
 namespace ondemand {

+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+simdjson_inline bool raw_json_string_is_quote_terminated(const uint8_t *json, uint32_t max_len) noexcept {
+  bool escaping{false};
+  for (uint32_t i = 1; i < max_len; i++) {
+    switch (json[i]) {
+      case '"':
+        if (!escaping) { return true; }
+        escaping = false;
+        break;
+      case '\\':
+        escaping = !escaping;
+        break;
+      default:
+        escaping = false;
+        break;
+    }
+  }
+  return false;
+}
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+
 simdjson_inline value_iterator::value_iterator(
   json_iterator *json_iter,
   depth_t depth,
@@ -156820,6 +200930,22 @@ simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iter
   return raw_json_string(key);
 }

+simdjson_warn_unused simdjson_inline error_code value_iterator::field_key_with_length(raw_json_string &key, std::size_t &len) noexcept {
+  assert_at_next();
+
+  const uint8_t *k = _json_iter->return_current_and_advance();
+  if (*(k++) != '"') { return report_error(TAPE_ERROR, "Object key is not a string"); }
+  // After return_current_and_advance(), the current token is the ':' that follows
+  // the key. The closing quote sits just before it (only JSON whitespace may
+  // intervene), so step back from the ':' to the closing quote to get the length.
+  // In minified JSON this is a single back-step.
+  const char *q = reinterpret_cast<const char *>(_json_iter->peek());
+  do { --q; } while (*q != '"');
+  key = raw_json_string(k);
+  len = static_cast<std::size_t>(q - reinterpret_cast<const char *>(k));
+  return SUCCESS;
+}
+
 simdjson_warn_unused simdjson_inline error_code value_iterator::field_value() noexcept {
   assert_at_next();

@@ -156937,7 +201063,7 @@ simdjson_warn_unused simdjson_inline error_code value_iterator::get_string(strin
   auto saved_string_buf_loc = _json_iter->string_buf_loc();
   auto err = get_string(allow_replacement).get(content);
   if (err) { return err; }
-  receiver = content;
+  internal::assign_utf8(receiver, content);
   // Restore the string buffer location, effectively discarding any temporary string storage
   _json_iter->string_buf_loc() = saved_string_buf_loc;
   return SUCCESS;
@@ -156948,6 +201074,15 @@ simdjson_warn_unused simdjson_inline simdjson_result<std::string_view> value_ite
 simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iterator::get_raw_json_string() noexcept {
   auto json = peek_scalar("string");
   if (*json != '"') { return incorrect_type_error("Not a string"); }
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  if (_json_iter->allow_incomplete_json()) {
+    const size_t remaining_input_length = _json_iter->remaining_input_length(json);
+    const uint32_t max_len = remaining_input_length < peek_start_length() ? uint32_t(remaining_input_length) : peek_start_length();
+    if (!raw_json_string_is_quote_terminated(json, max_len)) {
+      return STRING_ERROR;
+    }
+  }
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
   advance_scalar("string");
   return raw_json_string(json+1);
 }
@@ -156981,6 +201116,16 @@ simdjson_warn_unused simdjson_inline simdjson_result<double> value_iterator::get
   if(result.error() == SUCCESS) { advance_non_root_scalar("double"); }
   return result;
 }
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_float() noexcept {
+  auto result = numberparsing::parse_float(peek_non_root_scalar("float"));
+  if(result.error() == SUCCESS) { advance_non_root_scalar("float"); }
+  return result;
+}
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_float_in_string() noexcept {
+  auto result = numberparsing::parse_float_in_string(peek_non_root_scalar("float"));
+  if(result.error() == SUCCESS) { advance_non_root_scalar("float"); }
+  return result;
+}
 simdjson_warn_unused simdjson_inline simdjson_result<bool> value_iterator::get_bool() noexcept {
   auto result = parse_bool(peek_non_root_scalar("bool"));
   if(result.error() == SUCCESS) { advance_non_root_scalar("bool"); }
@@ -157083,7 +201228,7 @@ simdjson_warn_unused simdjson_inline error_code value_iterator::get_root_string(
   auto saved_string_buf_loc = _json_iter->string_buf_loc();
   auto err = get_root_string(check_trailing, allow_replacement).get(content);
   if (err) { return err; }
-  receiver = content;
+  internal::assign_utf8(receiver, content);
   // Restore the string buffer location, effectively discarding any temporary string storage
   _json_iter->string_buf_loc() = saved_string_buf_loc;
   return SUCCESS;
@@ -157095,6 +201240,15 @@ simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iter
   auto json = peek_scalar("string");
   if (*json != '"') { return incorrect_type_error("Not a string"); }
   if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  if (_json_iter->allow_incomplete_json()) {
+    const size_t remaining_input_length = _json_iter->remaining_input_length(json);
+    const uint32_t max_len = remaining_input_length < peek_root_length() ? uint32_t(remaining_input_length) : peek_root_length();
+    if (!raw_json_string_is_quote_terminated(json, max_len)) {
+      return STRING_ERROR;
+    }
+  }
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
   advance_scalar("string");
   return raw_json_string(json+1);
 }
@@ -157204,6 +201358,43 @@ simdjson_warn_unused simdjson_inline simdjson_result<double> value_iterator::get
   return result;
 }

+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_root_float(bool check_trailing) noexcept {
+  auto max_len = peek_root_length();
+  auto json = peek_root_scalar("float");
+  // We use the same buffer size as get_root_double: the number of significant
+  // digits that matter is smaller for binary32, but the JSON document may still
+  // spell out a long number that we must parse (and round) faithfully.
+  uint8_t tmpbuf[1074+8+1+1]; // +1 for null termination.
+  tmpbuf[1074+8+1] = '\0'; // make sure that buffer is always null terminated.
+  if (!_json_iter->copy_to_buffer(json, max_len, tmpbuf, 1074+8+1)) {
+    logger::log_error(*_json_iter, start_position(), depth(), "Root number more than 1082 characters");
+    return NUMBER_ERROR;
+  }
+  auto result = numberparsing::parse_float(tmpbuf);
+  if(result.error() == SUCCESS) {
+    if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+    advance_root_scalar("float");
+  }
+  return result;
+}
+
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_root_float_in_string(bool check_trailing) noexcept {
+  auto max_len = peek_root_length();
+  auto json = peek_root_scalar("float");
+  uint8_t tmpbuf[1074+8+1+1]; // +1 for null termination.
+  tmpbuf[1074+8+1] = '\0'; // make sure that buffer is always null terminated.
+  if (!_json_iter->copy_to_buffer(json, max_len, tmpbuf, 1074+8+1)) {
+    logger::log_error(*_json_iter, start_position(), depth(), "Root number more than 1082 characters");
+    return NUMBER_ERROR;
+  }
+  auto result = numberparsing::parse_float_in_string(tmpbuf);
+  if(result.error() == SUCCESS) {
+    if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+    advance_root_scalar("float");
+  }
+  return result;
+}
+
 simdjson_warn_unused simdjson_inline simdjson_result<bool> value_iterator::get_root_bool(bool check_trailing) noexcept {
   auto max_len = peek_root_length();
   auto json = peek_root_scalar("bool");
@@ -157432,6 +201623,21 @@ simdjson_inline void value_iterator::move_at_container_start() noexcept {
   _json_iter->token.set_position(_start_position + 1);
 }

+simdjson_inline void value_iterator::reenter_at(token_position position, depth_t depth) noexcept {
+  // Unlike reenter_child(), this does not require the live depth to be
+  // exactly one level shallower than depth, nor does it validate against
+  // the parser's per-depth container-start bookkeeping: neither holds in
+  // general for a caller-supplied snapshot (see object_position). What
+  // must still always hold, regardless of what was captured or how far
+  // the live iterator has since moved, is that position and depth are
+  // themselves sane values -- this is the same bound reenter_child()
+  // itself applies unconditionally.
+  SIMDJSON_ASSUME(position != nullptr);
+  SIMDJSON_ASSUME(depth >= 1 && depth < INT32_MAX);
+  _json_iter->_depth = depth;
+  _json_iter->token.set_position(position);
+}
+
 simdjson_inline simdjson_result<bool> value_iterator::reset_array() noexcept {
   if(error()) { return error(); }
   move_at_container_start();
@@ -158825,10 +203031,6 @@ simdjson_inline int count_ones(uint64_t input_num) {
   return __lasx_xvpickve2gr_w(__lasx_xvpcnt_d(__m256i(v4u64{input_num, 0, 0, 0})), 0);
 }

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2, uint64_t *result) {
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-}

 } // unnamed namespace
 } // namespace lasx
@@ -159382,7 +203584,7 @@ simdjson_inline escaping escaping::copy_and_find(const uint8_t *src, uint8_t *ds
 /* end file simdjson/lasx/begin.h */
 /* including simdjson/generic/ondemand/amalgamated.h for lasx: #include "simdjson/generic/ondemand/amalgamated.h" */
 /* begin file simdjson/generic/ondemand/amalgamated.h for lasx */
-#if defined(SIMDJSON_CONDITIONAL_INCLUDE) && !defined(SIMDJSON_GENERIC_BUILDER_DEPENDENCIES_H)
+#if defined(SIMDJSON_CONDITIONAL_INCLUDE) && !defined(SIMDJSON_GENERIC_ONDEMAND_DEPENDENCIES_H)
 #error simdjson/generic/ondemand/dependencies.h must be included before simdjson/generic/ondemand/amalgamated.h!
 #endif

@@ -159431,6 +203633,13 @@ class token_iterator;
 class value;
 class value_iterator;

+#if SIMDJSON_SUPPORTS_RANGES
+class array_range;
+class array_range_iterator;
+class object_range;
+class object_range_iterator;
+#endif // SIMDJSON_SUPPORTS_RANGES
+
 } // namespace ondemand
 } // namespace lasx
 } // namespace simdjson
@@ -159463,6 +203672,9 @@ template <> struct is_builtin_deserializable<lasx::ondemand::object> : std::true
 template <> struct is_builtin_deserializable<lasx::ondemand::value> : std::true_type {};
 template <> struct is_builtin_deserializable<lasx::ondemand::raw_json_string> : std::true_type {};
 template <> struct is_builtin_deserializable<std::string_view> : std::true_type {};
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template <> struct is_builtin_deserializable<std::u8string_view> : std::true_type {};
+#endif // SIMDJSON_SUPPORTS_CHAR8_T

 template <typename T>
 concept is_builtin_deserializable_v = is_builtin_deserializable<T>::value;
@@ -159480,6 +203692,10 @@ concept nothrow_custom_deserializable = nothrow_tag_invocable<deserialize_tag, V
 template <typename T, typename ValT = lasx::ondemand::value>
 concept nothrow_deserializable = nothrow_custom_deserializable<T, ValT> || is_builtin_deserializable_v<T>;

+// True iff get<T>() on ValT cannot throw: no custom tag_invoke, or that tag_invoke is noexcept.
+template <typename T, typename ValT = lasx::ondemand::value>
+concept nothrow_gettable = !custom_deserializable<T, ValT> || nothrow_custom_deserializable<T, ValT>;
+
 /// Deserialize Tag
 inline constexpr struct deserialize_tag {
   using array_type = lasx::ondemand::array;
@@ -159694,6 +203910,17 @@ public:
    */
   simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> field_key() noexcept;

+  /**
+   * Get the current field's key together with its raw byte length.
+   *
+   * Like field_key(), but also returns the number of raw key bytes (the distance
+   * from the first key byte to the closing quote). The length is recovered from
+   * the structural index -- the next structural token is the ':' -- by stepping
+   * back over any whitespace to the closing quote, avoiding a forward SIMD scan
+   * for the closing quote. Leaves the iterator positioned exactly as field_key().
+   */
+  simdjson_warn_unused simdjson_inline error_code field_key_with_length(raw_json_string &key, std::size_t &len) noexcept;
+
   /**
    * Pass the : in the field and move to its value.
    */
@@ -159846,6 +204073,8 @@ public:
   simdjson_warn_unused simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_warn_unused simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_warn_unused simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<bool> get_bool() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<bool> is_null() noexcept;
   simdjson_warn_unused simdjson_inline bool is_negative() noexcept;
@@ -159864,6 +204093,8 @@ public:
   simdjson_warn_unused simdjson_inline simdjson_result<int64_t> get_root_int64_in_string(bool check_trailing) noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<double> get_root_double(bool check_trailing) noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<double> get_root_double_in_string(bool check_trailing) noexcept;
+  simdjson_warn_unused simdjson_inline simdjson_result<float> get_root_float(bool check_trailing) noexcept;
+  simdjson_warn_unused simdjson_inline simdjson_result<float> get_root_float_in_string(bool check_trailing) noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<bool> get_root_bool(bool check_trailing) noexcept;
   simdjson_warn_unused simdjson_inline bool is_root_negative() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<bool> is_root_integer(bool check_trailing) noexcept;
@@ -159999,6 +204230,15 @@ protected:

   /** @copydoc error_code json_iterator::position() const noexcept; */
   simdjson_inline token_position position() const noexcept;
+  /**
+   * Move the live iterator directly to the given position and depth, without
+   * validating against the parser's per-depth container-start bookkeeping
+   * (unlike json_iterator::reenter_child()). Used to restore a previously
+   * captured mid-container position (see object::revert_position()): that
+   * bookkeeping only tracks each container's own start, not every position
+   * a caller might later capture and revert to, so it does not apply here.
+   */
+  simdjson_inline void reenter_at(token_position position, depth_t depth) noexcept;
   /** @copydoc error_code json_iterator::end_position() const noexcept; */
   simdjson_inline token_position last_position() const noexcept;
   /** @copydoc error_code json_iterator::end_position() const noexcept; */
@@ -160067,9 +204307,12 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+   * Supported types: object, array, raw_json_string, string_view, u8string_view (C++20), uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool
    *
-   * You may use get_double(), get_bool(), get_uint64(), get_int64(),
+   * You may use get_double(), get_float(), get_bool(), get_uint64(), get_int64(),
+   * get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(),
+   * get_float32(), get_float64(),
    * get_object(), get_array(), get_raw_json_string(), or get_string() instead.
    * When SIMDJSON_SUPPORTS_CONCEPTS is set, custom types are also supported.
    *
@@ -160079,7 +204322,7 @@ public:
   template <typename T>
   simdjson_inline simdjson_result<T> get()
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+    noexcept(nothrow_gettable<T, value>)
 #else
     noexcept
 #endif
@@ -160094,7 +204337,8 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+   * Supported types: object, array, raw_json_string, string_view, u8string_view (C++20), uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool
    * If the macro SIMDJSON_SUPPORTS_CONCEPTS is set, then custom types are also supported.
    *
    * @param out This is set to a value of the given type, parsed from the JSON. If there is an error, this may not be initialized.
@@ -160104,7 +204348,7 @@ public:
   template <typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out)
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+    noexcept(nothrow_gettable<T, value>)
 #else
     noexcept
 #endif
@@ -160132,7 +204376,7 @@ public:
       "And you do not seem to have added support for it. Indeed, we have that "
       "simdjson::custom_deserializable<T> is false and the type T is not a default type "
       "such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-      "int64_t, double, or bool.");
+      "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
     static_cast<void>(out); // to get rid of unused errors
     return UNINITIALIZED;
   }
@@ -160141,7 +204385,8 @@ public:
     // immediately fail.
     static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
       "The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-      "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+      "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+      "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
       " get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
       " You may also add support for custom types, see our documentation.");
     static_cast<void>(out); // to get rid of unused errors
@@ -160219,6 +204464,50 @@ public:
    */
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;

+  /**
+   * Cast this JSON value to a 16-bit unsigned integer.
+   *
+   * Calls get_uint64() and checks that the result fits in a uint16_t.
+   *
+   * @returns A 16-bit unsigned integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint16_t.
+   */
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+
+  /**
+   * Cast this JSON value to a 16-bit signed integer.
+   *
+   * Calls get_int64() and checks that the result fits in an int16_t.
+   *
+   * @returns A 16-bit signed integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int16_t.
+   */
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+
+  /**
+   * Cast this JSON value to an 8-bit unsigned integer.
+   *
+   * Calls get_uint64() and checks that the result fits in a uint8_t.
+   *
+   * @returns An 8-bit unsigned integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint8_t.
+   */
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+
+  /**
+   * Cast this JSON value to an 8-bit signed integer.
+   *
+   * Calls get_int64() and checks that the result fits in an int8_t.
+   *
+   * @returns An 8-bit signed integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int8_t.
+   */
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
+
   /**
    * Cast this JSON value to a double.
    *
@@ -160235,6 +204524,53 @@ public:
    */
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;

+  /**
+   * Cast this JSON value to a float (binary32).
+   *
+   * The value is rounded directly to binary32: it is the float nearest to the
+   * JSON number. Note that this may differ from get_double() followed by a cast
+   * to float, which rounds twice.
+   *
+   * @returns A float.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+
+  /**
+   * Cast this JSON value (inside string) to a float (binary32).
+   *
+   * @returns A float.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  /**
+   * Cast this JSON value to a std::float32_t (C++23).
+   *
+   * Same as get_float(): the value is rounded directly to binary32.
+   *
+   * @returns A std::float32_t.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  /**
+   * Cast this JSON value to a std::float64_t (C++23).
+   *
+   * Same as get_double().
+   *
+   * @returns A std::float64_t.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   */
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
+
   /**
    * Cast this JSON value to a string.
    *
@@ -160262,6 +204598,26 @@ public:
    */
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;

+#if SIMDJSON_SUPPORTS_CHAR8_T
+  /**
+   * Cast this JSON value to a C++20 UTF-8 string.
+   *
+   * The string is guaranteed to be valid UTF-8.
+   *
+   * Equivalent to get<std::u8string_view>(). This is a zero-copy alias of
+   * get_string(): the very same bytes are returned, viewed as char8_t.
+   *
+   * Important: a value should be consumed once. Calling get_u8string() twice on the same
+   * value is an error.
+   *
+   * @param allow_replacement Whether to allow a replacement character for unmatched surrogate pairs.
+   * @returns An UTF-8 string. The string is stored in the parser and will be invalidated the next
+   *          time it parses a document or when it is destroyed.
+   * @returns INCORRECT_TYPE if the JSON value is not a string.
+   */
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
   /**
    * Attempts to fill the provided std::string reference with the parsed value of the current string.
    *
@@ -160349,7 +204705,7 @@ public:
   /**
    * Cast this JSON value to an unsigned integer.
    *
-   * @returns A signed 64-bit integer.
+   * @returns A unsigned 64-bit integer.
    * @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit unsigned integer.
    */
   simdjson_inline operator uint64_t() noexcept(false);
@@ -160814,9 +205170,24 @@ public:
   simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template <typename string_type>
   simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -160824,9 +205195,19 @@ public:
   simdjson_inline simdjson_result<bool> get_bool() noexcept;
   simdjson_inline simdjson_result<bool> is_null() noexcept;

-  template<typename T> simdjson_inline simdjson_result<T> get() noexcept;
+  template<typename T> simdjson_inline simdjson_result<T> get()
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lasx::ondemand::value>);
+#else
+    noexcept;
+#endif

-  template<typename T> simdjson_inline error_code get(T &out) noexcept;
+  template<typename T> simdjson_inline error_code get(T &out)
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lasx::ondemand::value>);
+#else
+    noexcept;
+#endif

 #if SIMDJSON_EXCEPTIONS
   template <class T>
@@ -161157,6 +205538,7 @@ protected:
   token_position _position{};

   friend class json_iterator;
+  friend class document_stream;
   friend class value_iterator;
   friend class object;
   template <typename... Args>
@@ -161248,6 +205630,9 @@ protected:
    * value of this attribute.
    */
   bool _streaming{false};
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  bool _allow_incomplete_json{false};
+#endif

 public:
   simdjson_inline json_iterator() noexcept = default;
@@ -161272,6 +205657,10 @@ public:
    * start_root_array() and start_root_object().
    */
   simdjson_inline bool streaming() const noexcept;
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  simdjson_inline bool allow_incomplete_json() const noexcept;
+  simdjson_inline size_t remaining_input_length(const uint8_t *json) const noexcept;
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON

   /**
    * Get the root value iterator
@@ -162151,33 +206540,87 @@ public:
    * @param batch_size The batch size to use. MUST be larger than the largest document. The sweet
    *                   spot is cache-related: small enough to fit in cache, yet big enough to
    *                   parse as many documents as possible in one tight loop.
-   *                   Defaults to 10MB, which has been a reasonable sweet spot in our tests.
-   * @param allow_comma_separated (defaults on false) This allows a mode where the documents are
-   *                   separated by commas instead of whitespace. It comes with a performance
-   *                   penalty because the entire document is indexed at once (and the document must be
-   *                   less than 4 GB), and there is no multithreading. In this mode, the batch_size parameter
-   *                   is effectively ignored, as it is set to at least the document size.
+   *                   Defaults to 1MB, which has been a reasonable sweet spot in our tests.
+   * @param allow_comma_separated @deprecated Use stream_format::comma_delimited instead.
+   *                   When true, maps internally to stream_format::comma_delimited.
+   *                   Defaults to false.
    * @return The stream, or an error. An empty input will yield 0 documents rather than an EMPTY error. Errors:
    *         - MEMALLOC if the parser does not have enough capacity and memory allocation fails
    *         - CAPACITY if the parser does not have enough capacity and batch_size > max_capacity.
    *         - other json errors if parsing fails. You should not rely on these errors to always the same for the
    *           same document: they may vary under runtime dispatch (so they may vary depending on your system and hardware).
    */
-  inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
-  inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
-  inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
-  inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size)
+  inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size)
     the string might be automatically padded with up to SIMDJSON_PADDING whitespace characters */
-  inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
-  inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
+  inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @private An rvalue input is destroyed at the end of the full-expression, while the
+   * returned document_stream only holds a pointer to it: iterating the stream would then
+   * read freed memory. These deleted overloads also catch a std::string_view argument,
+   * which would otherwise convert implicitly to a padded_string temporary. */
+  inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size = DEFAULT_BATCH_SIZE) = delete;// unsafe
+  /** @private @overload iterate_many(const std::string &&s, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size = DEFAULT_BATCH_SIZE) = delete;// unsafe
   /** @private We do not want to allow implicit conversion from C string to std::string. */
   simdjson_result<document_stream> iterate_many(const char *buf, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept = delete;

+#ifndef SIMDJSON_DISABLE_DEPRECATED_API
+  /**
+   * @deprecated Use iterate_many with stream_format::comma_delimited instead.
+   */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @private Deleted for the same reason as iterate_many(const std::string &&s, size_t batch_size). */
+  inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size, bool allow_comma_separated) = delete;// unsafe
+  /** @private @overload iterate_many(const std::string &&s, size_t batch_size, bool allow_comma_separated) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size, bool allow_comma_separated) = delete;// unsafe
+#endif // SIMDJSON_DISABLE_DEPRECATED_API
+
+  /**
+   * Parse a stream of JSON documents with explicit format specification.
+   *
+   * @param buf The concatenated JSON documents.
+   * @param len The length of the buffer.
+   * @param batch_size The batch size to use.
+   * @param format The stream format (whitespace_delimited, json_sequence, or comma_delimited).
+   * @return A stream of documents, or an error.
+   */
+  inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(padded_string_view s, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format)
+   *
+   * The string is padded in place if needed (see simdjson::pad), as with iterate_many(std::string &s, size_t batch_size).
+   */
+  inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size, stream_format format) noexcept;
+  /** @private Deleted for the same reason as iterate_many(const std::string &&s, size_t batch_size). */
+  inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+  /** @private @overload iterate_many(const std::string &&s, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+
   /** The capacity of this parser (the largest document it can process). */
   simdjson_pure simdjson_inline size_t capacity() const noexcept;
   /** The maximum capacity of this parser (the largest document it is allowed to process). */
@@ -162305,6 +206748,7 @@ private:
   size_t _capacity{0};
   size_t _max_capacity;
   size_t _max_depth{DEFAULT_MAX_DEPTH};
+  size_t _document_len{0};
   std::unique_ptr<uint8_t[]> string_buf{};

 #if SIMDJSON_DEVELOPMENT_CHECKS
@@ -162367,8 +206811,19 @@ public:
    * Begin array iteration.
    *
    * Part of the std::iterable interface.
+   *
+   * With SIMDJSON_DEVELOPMENT_CHECKS, the iterator locks this array while it is
+   * alive, so that reentrant access (at(), count_elements(), reset(), ...) is
+   * reported as OUT_OF_ORDER_ITERATION.
    */
-  simdjson_inline simdjson_result<array_iterator> begin() noexcept;
+  simdjson_inline simdjson_result<array_iterator> begin() & noexcept;
+  /**
+   * Begin iteration over a temporary array, e.g., `v.get_array().begin()`.
+   *
+   * The iterator does not depend on the array instance and may outlive it, so
+   * it does not lock it.
+   */
+  simdjson_inline simdjson_result<array_iterator> begin() && noexcept;
   /**
    * Sentinel representing the end of the array.
    *
@@ -162499,7 +206954,7 @@ public:
    */
   template <typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out)
-     noexcept(custom_deserializable<T, array> ? nothrow_custom_deserializable<T, array> : true) {
+     noexcept(nothrow_gettable<T, array>) {
     static_assert(custom_deserializable<T, array>);
     return deserialize(*this, out);
   }
@@ -162511,7 +206966,7 @@ public:
    */
   template <typename T>
   simdjson_inline simdjson_result<T> get()
-    noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+    noexcept(nothrow_gettable<T, array>)
   {
     static_assert(std::is_default_constructible<T>::value, "The specified type is not default constructible.");
     T out{};
@@ -162567,6 +207022,10 @@ protected:
    * iter.is_alive() == false indicates iteration is complete.
    */
   value_iterator iter{};
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  bool locked{false};
+  simdjson_inline void set_locked(bool _locked) noexcept;
+#endif

   friend class value;
   friend class document;
@@ -162588,7 +207047,8 @@ public:
   simdjson_inline simdjson_result(error_code error) noexcept; ///< @private
   simdjson_inline simdjson_result() noexcept = default;

-  simdjson_inline simdjson_result<lasx::ondemand::array_iterator> begin() noexcept;
+  simdjson_inline simdjson_result<lasx::ondemand::array_iterator> begin() & noexcept;
+  simdjson_inline simdjson_result<lasx::ondemand::array_iterator> begin() && noexcept;
   simdjson_inline simdjson_result<lasx::ondemand::array_iterator> end() noexcept;
   inline simdjson_result<size_t> count_elements() & noexcept;
   inline simdjson_result<bool> is_empty() & noexcept;
@@ -162608,7 +207068,7 @@ public:
   // TODO: move this code into object-inl.h

   template<typename T>
-  simdjson_inline simdjson_result<T> get() noexcept {
+  simdjson_inline simdjson_result<T> get() noexcept(nothrow_gettable<T, lasx::ondemand::array>) {
     if (error()) { return error(); }
     if constexpr (std::is_same_v<T, lasx::ondemand::array>) {
       return first;
@@ -162616,7 +207076,7 @@ public:
     return first.get<T>();
   }
   template<typename T>
-  simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept {
+  simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept(nothrow_gettable<T, lasx::ondemand::array>) {
     if (error()) { return error(); }
     if constexpr (std::is_same_v<T, lasx::ondemand::array>) {
       out = first;
@@ -162668,6 +207128,15 @@ public:
   /** Create a new, invalid array iterator. */
   simdjson_inline array_iterator() noexcept = default;

+#if SIMDJSON_DEVELOPMENT_CHECKS
+  simdjson_inline ~array_iterator() noexcept;
+
+  simdjson_inline array_iterator(array_iterator&&) noexcept;
+  simdjson_inline array_iterator& operator=(array_iterator&&) noexcept;
+  simdjson_inline array_iterator(const array_iterator&) noexcept;
+  simdjson_inline array_iterator& operator=(const array_iterator&) noexcept;
+#endif
+
   //
   // Iterator interface
   //
@@ -162710,6 +207179,9 @@ public:
 private:
 #if SIMDJSON_DEVELOPMENT_CHECKS
    bool has_been_referenced{false};
+   array* parent{nullptr};
+
+   simdjson_inline array_iterator(const value_iterator &_iter, array* _parent) noexcept;
 #endif
   value_iterator iter{};

@@ -162809,14 +207281,14 @@ public:
   /**
    * Cast this JSON value to an unsigned integer.
    *
-   * @returns A signed 64-bit integer.
+   * @returns A unsigned 64-bit integer.
    * @returns INCORRECT_TYPE If the JSON value is not a 64-bit unsigned integer.
    */
   simdjson_inline simdjson_result<uint64_t> get_uint64() noexcept;
   /**
    * Cast this JSON value (inside string) to an unsigned integer.
    *
-   * @returns A signed 64-bit integer.
+   * @returns A unsigned 64-bit integer.
    * @returns INCORRECT_TYPE If the JSON value is not a 64-bit unsigned integer.
    */
   simdjson_inline simdjson_result<uint64_t> get_uint64_in_string() noexcept;
@@ -162854,6 +207326,46 @@ public:
    * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int32_t.
    */
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  /**
+   * Cast this JSON value to a 16-bit unsigned integer.
+   *
+   * Calls get_uint64() and checks that the result fits in a uint16_t.
+   *
+   * @returns A 16-bit unsigned integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint16_t.
+   */
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  /**
+   * Cast this JSON value to a 16-bit signed integer.
+   *
+   * Calls get_int64() and checks that the result fits in an int16_t.
+   *
+   * @returns A 16-bit signed integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int16_t.
+   */
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  /**
+   * Cast this JSON value to an 8-bit unsigned integer.
+   *
+   * Calls get_uint64() and checks that the result fits in a uint8_t.
+   *
+   * @returns An 8-bit unsigned integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint8_t.
+   */
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  /**
+   * Cast this JSON value to an 8-bit signed integer.
+   *
+   * Calls get_int64() and checks that the result fits in an int8_t.
+   *
+   * @returns An 8-bit signed integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int8_t.
+   */
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   /**
    * Cast this JSON value to a double.
    *
@@ -162869,6 +207381,53 @@ public:
    * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
    */
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+
+  /**
+   * Cast this JSON value to a float (binary32).
+   *
+   * The value is rounded directly to binary32: it is the float nearest to the
+   * JSON number. Note that this may differ from get_double() followed by a cast
+   * to float, which rounds twice.
+   *
+   * @returns A float.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+
+  /**
+   * Cast this JSON value (inside string) to a float (binary32).
+   *
+   * @returns A float.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  /**
+   * Cast this JSON value to a std::float32_t (C++23).
+   *
+   * Same as get_float(): the value is rounded directly to binary32.
+   *
+   * @returns A std::float32_t.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  /**
+   * Cast this JSON value to a std::float64_t (C++23).
+   *
+   * Same as get_double().
+   *
+   * @returns A std::float64_t.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   */
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   /**
    * Cast this JSON value to a string.
    *
@@ -162882,6 +207441,24 @@ public:
    * @returns INCORRECT_TYPE if the JSON value is not a string.
    */
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  /**
+   * Cast this JSON value to a C++20 UTF-8 string.
+   *
+   * The string is guaranteed to be valid UTF-8.
+   *
+   * Equivalent to get<std::u8string_view>(). This is a zero-copy alias of
+   * get_string(): the very same bytes are returned, viewed as char8_t.
+   *
+   * Important: Calling get_u8string() twice on the same document is an error.
+   *
+   * @param allow_replacement Whether to allow a replacement character for unmatched surrogate pairs.
+   * @returns An UTF-8 string. The string is stored in the parser and will be invalidated the next
+   *          time it parses a document or when it is destroyed.
+   * @returns INCORRECT_TYPE if the JSON value is not a string.
+   */
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   /**
    * Attempts to fill the provided std::string reference with the parsed value of the current string.
    *
@@ -162952,9 +207529,12 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool
    *
-   * You may use get_double(), get_bool(), get_uint64(), get_int64(),
+   * You may use get_double(), get_float(), get_bool(), get_uint64(), get_int64(),
+   * get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(),
+   * get_float32(), get_float64(),
    * get_object(), get_array(), get_raw_json_string(), or get_string() instead.
    *
    * @returns A value of the given type, parsed from the JSON.
@@ -162963,7 +207543,7 @@ public:
   template <typename T>
   simdjson_inline simdjson_result<T> get() &
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    noexcept(nothrow_gettable<T, document>)
 #else
     noexcept
 #endif
@@ -162986,7 +207566,7 @@ public:
   template<typename T>
   simdjson_inline simdjson_result<T> get() &&
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    noexcept(nothrow_gettable<T, document>)
 #else
     noexcept
 #endif
@@ -162998,7 +207578,8 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool, value
+   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool, value
    *
    * Be mindful that the document instance must remain in scope while you are accessing object, array and value instances.
    *
@@ -163009,7 +207590,7 @@ public:
   template<typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out) &
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    noexcept(nothrow_gettable<T, document>)
 #else
     noexcept
 #endif
@@ -163022,7 +207603,7 @@ public:
         "And you do not seem to have added support for it. Indeed, we have that "
         "simdjson::custom_deserializable<T> is false and the type T is not a default type "
         "such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-        "int64_t, double, or bool.");
+        "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
       static_cast<void>(out); // to get rid of unused errors
       return UNINITIALIZED;
     }
@@ -163031,7 +207612,8 @@ public:
     // immediately fail.
     static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
       "The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-      "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+      "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+      "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
       " get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
       " You may also add support for custom types, see our documentation.");
     static_cast<void>(out); // to get rid of unused errors
@@ -163040,7 +207622,12 @@ public:
   }

   /** @overload template<typename T> error_code get(T &out) & noexcept */
-  template<typename T> simdjson_deprecated simdjson_inline error_code get(T &out) && noexcept;
+  template<typename T> simdjson_deprecated simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, document>);
+#else
+    noexcept;
+#endif

 #if SIMDJSON_EXCEPTIONS
   /**
@@ -163074,24 +207661,24 @@ public:
   /**
    * Cast this JSON value to an unsigned integer.
    *
-   * @returns A signed 64-bit integer.
+   * @returns A unsigned 64-bit integer.
    * @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit unsigned integer.
    */
-  simdjson_inline operator uint64_t() noexcept(false);
+  explicit simdjson_inline operator uint64_t() noexcept(false);
   /**
    * Cast this JSON value to a signed integer.
    *
    * @returns A signed 64-bit integer.
    * @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit integer.
    */
-  simdjson_inline operator int64_t() noexcept(false);
+  explicit simdjson_inline operator int64_t() noexcept(false);
   /**
    * Cast this JSON value to a double.
    *
    * @returns A double.
    * @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a valid floating-point number.
    */
-  simdjson_inline operator double() noexcept(false);
+  explicit simdjson_inline operator double() noexcept(false);
   /**
    * Cast this JSON value to a string.
    *
@@ -163101,7 +207688,7 @@ public:
    *          time it parses a document or when it is destroyed.
    * @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not a string.
    */
-  simdjson_inline operator std::string_view() noexcept(false) simdjson_lifetime_bound;
+  explicit simdjson_inline operator std::string_view() noexcept(false) simdjson_lifetime_bound;
   /**
    * Cast this JSON value to a raw_json_string.
    *
@@ -163110,14 +207697,14 @@ public:
    * @returns A pointer to the raw JSON for the given string.
    * @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not a string.
    */
-  simdjson_inline operator raw_json_string() noexcept(false) simdjson_lifetime_bound;
+  explicit simdjson_inline operator raw_json_string() noexcept(false) simdjson_lifetime_bound;
   /**
    * Cast this JSON value to a bool.
    *
    * @returns A bool value.
    * @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not true or false.
    */
-  simdjson_inline operator bool() noexcept(false);
+  explicit simdjson_inline operator bool() noexcept(false);
   /**
    * Cast this JSON value to a value when the document is an object or an array.
    *
@@ -163612,9 +208199,24 @@ public:
   simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template <typename string_type>
   simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -163626,7 +208228,7 @@ public:
   template <typename T>
   simdjson_inline simdjson_result<T> get() &
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    noexcept(nothrow_gettable<T, document_reference>)
 #else
     noexcept
 #endif
@@ -163639,7 +208241,8 @@ public:
   template<typename T>
   simdjson_inline simdjson_result<T> get() &&
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    // Forwards to document::get<T>(), so the document customization decides.
+    noexcept(nothrow_gettable<T, document>)
 #else
     noexcept
 #endif
@@ -163651,7 +208254,8 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool, value
+   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool, value
    *
    * Be mindful that the document instance must remain in scope while you are accessing object, array and value instances.
    *
@@ -163662,7 +208266,7 @@ public:
   template<typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out) &
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document_reference> : true)
+    noexcept(nothrow_gettable<T, document_reference>)
 #else
     noexcept
 #endif
@@ -163675,7 +208279,7 @@ public:
         "And you do not seem to have added support for it. Indeed, we have that "
         "simdjson::custom_deserializable<T> is false and the type T is not a default type "
         "such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-        "int64_t, double, or bool.");
+        "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
       static_cast<void>(out); // to get rid of unused errors
       return UNINITIALIZED;
     }
@@ -163684,7 +208288,8 @@ public:
     // immediately fail.
     static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
       "The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-      "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+      "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+      "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
       " get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
       " You may also add support for custom types, see our documentation.");
     static_cast<void>(out); // to get rid of unused errors
@@ -163693,7 +208298,12 @@ public:
   }

   /** @overload template<typename T> error_code get(T &out) & noexcept */
-  template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+  template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, document_reference>);
+#else
+    noexcept;
+#endif
   simdjson_inline simdjson_result<std::string_view> raw_json() noexcept;
 #if SIMDJSON_STATIC_REFLECTION
   template<constevalutil::fixed_string... FieldNames, typename T>
@@ -163706,12 +208316,12 @@ public:
   explicit simdjson_inline operator T() noexcept(false);
   simdjson_inline operator array() & noexcept(false);
   simdjson_inline operator object() & noexcept(false);
-  simdjson_inline operator uint64_t() noexcept(false);
-  simdjson_inline operator int64_t() noexcept(false);
-  simdjson_inline operator double() noexcept(false);
-  simdjson_inline operator std::string_view() noexcept(false);
-  simdjson_inline operator raw_json_string() noexcept(false);
-  simdjson_inline operator bool() noexcept(false);
+  explicit simdjson_inline operator uint64_t() noexcept(false);
+  explicit simdjson_inline operator int64_t() noexcept(false);
+  explicit simdjson_inline operator double() noexcept(false);
+  explicit simdjson_inline operator std::string_view() noexcept(false);
+  explicit simdjson_inline operator raw_json_string() noexcept(false);
+  explicit simdjson_inline operator bool() noexcept(false);
   simdjson_inline operator value() noexcept(false);
 #endif
   simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -163773,9 +208383,24 @@ public:
   simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template <typename string_type>
   simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -163784,11 +208409,31 @@ public:
   simdjson_inline simdjson_result<lasx::ondemand::value> get_value() noexcept;
   simdjson_inline simdjson_result<bool> is_null() noexcept;

-  template<typename T> simdjson_inline simdjson_result<T> get() & noexcept;
-  template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() && noexcept;
+  template<typename T> simdjson_inline simdjson_result<T> get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lasx::ondemand::document>);
+#else
+    noexcept;
+#endif
+  template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lasx::ondemand::document>);
+#else
+    noexcept;
+#endif

-  template<typename T> simdjson_inline error_code get(T &out) & noexcept;
-  template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+  template<typename T> simdjson_inline error_code get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lasx::ondemand::document>);
+#else
+    noexcept;
+#endif
+  template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lasx::ondemand::document>);
+#else
+    noexcept;
+#endif
 #if SIMDJSON_EXCEPTIONS

   using lasx::implementation_simdjson_result_base<lasx::ondemand::document>::operator*;
@@ -163797,12 +208442,12 @@ public:
   explicit simdjson_inline operator T() noexcept(false);
   simdjson_inline operator lasx::ondemand::array() & noexcept(false);
   simdjson_inline operator lasx::ondemand::object() & noexcept(false);
-  simdjson_inline operator uint64_t() noexcept(false);
-  simdjson_inline operator int64_t() noexcept(false);
-  simdjson_inline operator double() noexcept(false);
-  simdjson_inline operator std::string_view() noexcept(false);
-  simdjson_inline operator lasx::ondemand::raw_json_string() noexcept(false);
-  simdjson_inline operator bool() noexcept(false);
+  explicit simdjson_inline operator uint64_t() noexcept(false);
+  explicit simdjson_inline operator int64_t() noexcept(false);
+  explicit simdjson_inline operator double() noexcept(false);
+  explicit simdjson_inline operator std::string_view() noexcept(false);
+  explicit simdjson_inline operator lasx::ondemand::raw_json_string() noexcept(false);
+  explicit simdjson_inline operator bool() noexcept(false);
   simdjson_inline operator lasx::ondemand::value() noexcept(false);
 #endif
   simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -163868,9 +208513,24 @@ public:
   simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template <typename string_type>
   simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -163879,22 +208539,42 @@ public:
   simdjson_inline simdjson_result<lasx::ondemand::value> get_value() noexcept;
   simdjson_inline simdjson_result<bool> is_null() noexcept;

-  template<typename T> simdjson_inline simdjson_result<T> get() & noexcept;
-  template<typename T> simdjson_inline simdjson_result<T> get() && noexcept;
+  template<typename T> simdjson_inline simdjson_result<T> get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lasx::ondemand::document_reference>);
+#else
+    noexcept;
+#endif
+  template<typename T> simdjson_inline simdjson_result<T> get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lasx::ondemand::document_reference>);
+#else
+    noexcept;
+#endif

-  template<typename T> simdjson_inline error_code get(T &out) & noexcept;
-  template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+  template<typename T> simdjson_inline error_code get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lasx::ondemand::document_reference>);
+#else
+    noexcept;
+#endif
+  template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lasx::ondemand::document_reference>);
+#else
+    noexcept;
+#endif
 #if SIMDJSON_EXCEPTIONS
   template <class T>
   explicit simdjson_inline operator T() noexcept(false);
   simdjson_inline operator lasx::ondemand::array() & noexcept(false);
   simdjson_inline operator lasx::ondemand::object() & noexcept(false);
-  simdjson_inline operator uint64_t() noexcept(false);
-  simdjson_inline operator int64_t() noexcept(false);
-  simdjson_inline operator double() noexcept(false);
-  simdjson_inline operator std::string_view() noexcept(false);
-  simdjson_inline operator lasx::ondemand::raw_json_string() noexcept(false);
-  simdjson_inline operator bool() noexcept(false);
+  explicit simdjson_inline operator uint64_t() noexcept(false);
+  explicit simdjson_inline operator int64_t() noexcept(false);
+  explicit simdjson_inline operator double() noexcept(false);
+  explicit simdjson_inline operator std::string_view() noexcept(false);
+  explicit simdjson_inline operator lasx::ondemand::raw_json_string() noexcept(false);
+  explicit simdjson_inline operator bool() noexcept(false);
   simdjson_inline operator lasx::ondemand::value() noexcept(false);
 #endif
   simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -164062,10 +208742,7 @@ public:
    *   }
    *   size_t truncated = stream.truncated_bytes();
    *
-   * IMPORTANT: this value is only meaningful under the conditions below. It is
-   * computed from stage-1 bookkeeping, and outside these conditions it is not
-   * merely imprecise, it is arbitrary -- it can exceed size_in_bytes() or wrap
-   * around to a huge value. Check it only when both of the following hold:
+   * IMPORTANT: this value is only meaningful under the conditions below.
    *
    *   - you iterated all the way to the end of the stream;
    *   - no document reported an error. Iteration stops at the first failed
@@ -164074,6 +208751,9 @@ public:
    * If you need to know about a truncated tail outside those conditions, track
    * it yourself from the last successful document (see iterator::current_index()
    * and iterator::source()).
+   *
+   * An empty input (zero bytes) or an input made only of white space contains
+   * no document: truncated_bytes() returns zero.
    */
   inline size_t truncated_bytes() const noexcept;

@@ -164133,7 +208813,10 @@ public:
      *
      * The returned string_view instance is simply a map to the (unparsed)
      * source string: it may thus include white-space characters and all manner
-     * of padding.
+     * of padding. It spans the whole current document, whether or not you
+     * have already accessed (part of) the document. Thus
+     * current_index() + source().size() is the offset just past the end of the
+     * current document, which is useful when reading a stream in chunks.
      *
      * This function (source()) is experimental and the usage
      * may change in future versions of simdjson: we find the API somewhat
@@ -164187,13 +208870,16 @@ private:
    * @param buf is the raw byte buffer we need to process
    * @param len is the length of the raw byte buffer in bytes
    * @param batch_size is the size of the windows (must be strictly greater or equal to the largest JSON document)
+   * @param allow_comma_separated whether to allow comma-separated documents
+   * @param format the stream format
    */
   simdjson_inline document_stream(
     ondemand::parser &parser,
     const uint8_t *buf,
     size_t len,
     size_t batch_size,
-    bool allow_comma_separated
+    bool allow_comma_separated,
+    stream_format format = stream_format::whitespace_delimited
   ) noexcept;

   /**
@@ -164227,8 +208913,23 @@ private:
    */
   inline void next() noexcept;

-  /** Move the json_iterator of the document to the location of the next document in the stream. */
+  /**
+   * Move the json_iterator of the document to the location of the next document
+   * in the stream.
+   *
+   * For formats with a document delimiter (`newline_delimited`, `json_sequence`),
+   * when the iterator is still inside the current document (`depth() > 0`), this
+   * may jump to the next delimiter instead of walking remaining structurals. That
+   * jump does not structure-validate the unread remainder.
+   */
   inline void next_document() noexcept;
+  /** Byte that ends a document under `format`, or 0 if there is none. */
+  simdjson_inline uint8_t document_delimiter() const noexcept;
+  /**
+   * Position the iterator at the first structural at or past the next
+   * `delimiter` in the current batch. Returns false if none is found.
+   */
+  simdjson_inline bool skip_to_delimiter(uint8_t delimiter) noexcept;

   /** Get the next document index. */
   inline size_t next_batch_start() const noexcept;
@@ -164242,6 +208943,7 @@ private:
   size_t len;
   size_t batch_size;
   bool allow_comma_separated;
+  stream_format format;
   /**
    * We are going to use just one document instance. The document owns
    * the json_iterator. It implies that we only ever pass a reference
@@ -164268,7 +208970,7 @@ private:
   /** The error returned from the stage 1 thread. */
   error_code stage1_thread_error{UNINITIALIZED};
   /** The thread used to run stage 1 against the next batch in the background. */
-  std::unique_ptr<stage1_worker> worker{new(std::nothrow) stage1_worker()};
+  std::unique_ptr<stage1_worker> worker{};
   /**
    * The parser used to run stage 1 in the background. Will be swapped
    * with the regular parser when finished.
@@ -164343,6 +209045,16 @@ public:
    * call it again nor can you call key().
    */
   simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  /**
+   * Get the key as a C++20 u8string_view. This is a zero-copy alias of
+   * unescaped_key(): the very same bytes are returned, viewed as char8_t.
+   *
+   * This consumes the key: once you have called unescaped_u8key(), you cannot
+   * call it again nor can you call key().
+   */
+  simdjson_inline simdjson_warn_unused simdjson_result<std::u8string_view> unescaped_u8key(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   /**
    * Get the key as a string_view (for higher speed, consider raw_key).
    * We deliberately use a more cumbersome name (unescaped_key) to force users
@@ -164375,6 +209087,16 @@ public:
    * you can safely call it repeatedly. Use unescaped_key() to get the unescaped key.
    */
   simdjson_inline std::string_view escaped_key() const noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  /**
+   * Get the key as a C++20 u8string_view. This is a zero-copy alias of
+   * escaped_key(): the very same bytes are returned, viewed as char8_t.
+   * The string is unprocessed, so it may contain escape characters
+   * (e.g., \uXXXX or \n). It does not count as a consumption of the content:
+   * you can safely call it repeatedly.
+   */
+  simdjson_inline std::u8string_view escaped_u8key() const noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   /**
    * Get the field value.
    */
@@ -164406,11 +209128,17 @@ public:
   simdjson_inline simdjson_result() noexcept = default;

   simdjson_inline simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> unescaped_u8key(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template<typename string_type>
   simdjson_inline error_code unescaped_key(string_type &receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<lasx::ondemand::raw_json_string> key() noexcept;
   simdjson_inline simdjson_result<std::string_view> key_raw_json_token() noexcept;
   simdjson_inline simdjson_result<std::string_view> escaped_key() noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> escaped_u8key() noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   simdjson_inline simdjson_result<lasx::ondemand::value> value() noexcept;
 };

@@ -164418,6 +209146,1398 @@ public:

 #endif // SIMDJSON_GENERIC_ONDEMAND_FIELD_H
 /* end file simdjson/generic/ondemand/field.h for lasx */
+/* including simdjson/generic/ondemand/key_selector.h for lasx: #include "simdjson/generic/ondemand/key_selector.h" */
+/* begin file simdjson/generic/ondemand/key_selector.h for lasx */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+#define SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #include "simdjson/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/common_defs.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/constevalutil.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/raw_json_string.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#include <array>
+#include <string>      // std::string (key_selector::describe)
+#include <string_view>
+#include <cstddef>
+#include <cstdint>
+#include <cstring>     // std::memcpy (portable unaligned window load)
+#include <utility>     // std::index_sequence (window candidate dispatch)
+#include <type_traits> // std::integral_constant (window candidate dispatch)
+
+// AArch64 only: 32-bit ARM (ARMv7) defines __ARM_NEON too, but lacks the
+// A64-only horizontal reductions (vmaxvq_u32) used below. See issue #2885.
+#if defined(__aarch64__)
+  #include <arm_neon.h>
+  #define SIMDJSON_KEY_SELECTOR_HAS_NEON 1
+#else
+  #define SIMDJSON_KEY_SELECTOR_HAS_NEON 0
+#endif
+#if defined(__SSE2__)
+  #include <emmintrin.h>
+  #define SIMDJSON_KEY_SELECTOR_HAS_SSE2 1
+#else
+  #define SIMDJSON_KEY_SELECTOR_HAS_SSE2 0
+#endif
+#if defined(__loongarch_sx)
+  #include <lsxintrin.h>
+  #define SIMDJSON_KEY_SELECTOR_HAS_LSX 1
+#else
+  #define SIMDJSON_KEY_SELECTOR_HAS_LSX 0
+#endif
+
+#if SIMDJSON_SUPPORTS_CONCEPTS
+
+namespace simdjson {
+namespace lasx {
+namespace ondemand {
+
+namespace key_selector_detail {
+
+// Since not constexpr, triggers compile-time error.
+inline void compile_time_error(const char* message) noexcept { (void)message; }
+
+// ============================================================================
+// Compile-time perfect-hash generator.
+// It scales to ~100 keys at compile time by determining association values one (position, character)
+// symbol at a time (gperf-style) instead of an exhaustive offset search, and
+// falls back to a Hash-and-Displace construction for large/awkward key sets.
+//
+// Only flat tables survive to runtime; the lookup is a few additions plus a
+// single SIMD key comparison (see match_raw below).
+// ============================================================================
+
+// Maximum number of character positions the gperf hash may combine.
+static constexpr std::size_t MAX_POSITIONS = 16;
+// Sentinel "position" meaning "the last character of the key".
+static constexpr std::size_t LAST_CHAR = std::size_t(-1);
+// Runtime-encoded sentinels (stored in uint8 tables).
+static constexpr std::uint8_t POS_LAST_CHAR = 0xFF; // positions_[i] == last char
+static constexpr std::uint8_t HD_MODE       = 0xFF; // num_positions == H&D mode
+// Flags stored in positions[2] in H&D mode to select the key-hash variant.
+static constexpr std::size_t HD_HASH_2BYTE_FLAG = 2;
+static constexpr std::size_t HD_HASH_4BYTE_FLAG = 4;
+
+constexpr std::size_t next_power_of_2(std::size_t n) noexcept {
+    if (n == 0) { return 1; }
+    std::size_t p = 1;
+    while (p < n) { p <<= 1; }
+    return p;
+}
+
+// Character at a given position (LAST_CHAR means last character), or 256 if out
+// of bounds.
+constexpr std::size_t char_at(std::string_view key, std::size_t pos) noexcept {
+    if (pos == LAST_CHAR) {
+        if (key.empty()) { return 256; }
+        return static_cast<unsigned char>(key[key.size() - 1]);
+    }
+    if (pos >= key.size()) { return 256; }
+    return static_cast<unsigned char>(key[pos]);
+}
+
+// Count key pairs that a set of positions fails to distinguish. Keys whose
+// lengths differ modulo the table size are separated by the length term in the
+// hash, so they need no position coverage.
+template <std::size_t N>
+consteval std::size_t count_undistinguished_pairs(
+    const std::array<std::string_view, N>& keys,
+    const std::size_t* positions,
+    std::size_t num_positions,
+    std::size_t modulus) {
+    std::size_t count = 0;
+    for (std::size_t i = 0; i < N; ++i) {
+        for (std::size_t j = i + 1; j < N; ++j) {
+            if (keys[i].size() % modulus != keys[j].size() % modulus) { continue; }
+            bool distinguished = false;
+            for (std::size_t p = 0; p < num_positions; ++p) {
+                if (char_at(keys[i], positions[p]) != char_at(keys[j], positions[p])) {
+                    distinguished = true;
+                    break;
+                }
+            }
+            if (!distinguished) { ++count; }
+        }
+    }
+    return count;
+}
+
+template <std::size_t N>
+consteval bool positions_distinguish(
+    const std::array<std::string_view, N>& keys,
+    const std::size_t* positions,
+    std::size_t num_positions,
+    std::size_t modulus) {
+    return count_undistinguished_pairs<N>(keys, positions, num_positions, modulus) == 0;
+}
+
+// Number of distinct (length % modulus, char_at(key, pos)) pairs at a position.
+template <std::size_t N>
+consteval std::size_t discriminating_power(
+    const std::array<std::string_view, N>& keys,
+    std::size_t pos,
+    std::size_t modulus) {
+    struct pair { std::size_t len_mod; std::size_t ch; };
+    std::array<pair, N> pairs{};
+    for (std::size_t i = 0; i < N; ++i) {
+        pairs[i] = {keys[i].size() % modulus, char_at(keys[i], pos)};
+    }
+    std::size_t count = 0;
+    for (std::size_t i = 0; i < N; ++i) {
+        bool dup = false;
+        for (std::size_t j = 0; j < i; ++j) {
+            if (pairs[i].len_mod == pairs[j].len_mod && pairs[i].ch == pairs[j].ch) {
+                dup = true;
+                break;
+            }
+        }
+        if (!dup) { ++count; }
+    }
+    return count;
+}
+
+template <std::size_t N>
+consteval std::size_t max_key_length(const std::array<std::string_view, N>& keys) {
+    std::size_t m = 0;
+    for (std::size_t i = 0; i < N; ++i) {
+        if (keys[i].size() > m) { m = keys[i].size(); }
+    }
+    return m;
+}
+
+// Bounded backtracking DFS for a minimal set of distinguishing positions.
+template <std::size_t N>
+consteval bool backtracking_search(
+    const std::array<std::string_view, N>& keys,
+    const std::size_t* candidates,
+    std::size_t num_candidates,
+    std::size_t* positions,
+    std::size_t& num_positions_out,
+    std::size_t& budget,
+    std::size_t modulus) {
+    constexpr std::size_t MAX_DEPTH = 8;
+    std::size_t breadth = num_candidates < 20 ? num_candidates : 20;
+
+    struct frame { std::size_t depth; std::size_t next_ci; std::size_t parent_count; };
+    std::array<frame, MAX_DEPTH + 1> stack{};
+    std::size_t sp = 0;
+
+    std::size_t initial_count = count_undistinguished_pairs<N>(keys, positions, 0, modulus);
+    if (budget > 0) { --budget; }
+    if (initial_count == 0) { num_positions_out = 0; return true; }
+
+    stack[0] = {0, 0, initial_count};
+
+    while (budget > 0) {
+        if (sp > MAX_DEPTH) {
+            if (sp == 0) { break; }
+            --sp;
+            ++stack[sp].next_ci;
+            continue;
+        }
+        auto& f = stack[sp];
+        if (f.next_ci >= breadth) {
+            if (sp == 0) { break; }
+            --sp;
+            ++stack[sp].next_ci;
+            continue;
+        }
+        positions[sp] = candidates[f.next_ci];
+        --budget;
+        std::size_t new_count = count_undistinguished_pairs<N>(keys, positions, sp + 1, modulus);
+        if (new_count == 0) { num_positions_out = sp + 1; return true; }
+        if (new_count < f.parent_count && sp + 1 < MAX_DEPTH) {
+            stack[sp + 1] = {sp + 1, f.next_ci + 1, new_count};
+            ++sp;
+        } else {
+            ++f.next_ci;
+        }
+    }
+    return false;
+}
+
+// Phase 1: select character positions that distinguish all colliding pairs.
+template <std::size_t N>
+consteval std::size_t select_positions(
+    const std::array<std::string_view, N>& keys,
+    std::array<std::size_t, MAX_POSITIONS>& positions,
+    std::size_t modulus) {
+    if (positions_distinguish<N>(keys, positions.data(), 0, modulus)) { return 0; }
+
+    std::size_t max_len = max_key_length(keys);
+    constexpr std::size_t MAX_CANDIDATES = 256;
+    std::array<std::size_t, MAX_CANDIDATES> candidates{};
+    std::array<std::size_t, MAX_CANDIDATES> powers{};
+    std::size_t num_candidates = 0;
+    for (std::size_t p = 0; p < max_len && num_candidates < MAX_CANDIDATES - 1; ++p) {
+        candidates[num_candidates] = p;
+        powers[num_candidates] = discriminating_power(keys, p, modulus);
+        ++num_candidates;
+    }
+    if (num_candidates < MAX_CANDIDATES) {
+        candidates[num_candidates] = LAST_CHAR;
+        powers[num_candidates] = discriminating_power(keys, LAST_CHAR, modulus);
+        ++num_candidates;
+    }
+    for (std::size_t i = 0; i < num_candidates; ++i) {
+        for (std::size_t j = i + 1; j < num_candidates; ++j) {
+            if (powers[j] > powers[i]) {
+                auto tc = candidates[i]; candidates[i] = candidates[j]; candidates[j] = tc;
+                auto tp = powers[i]; powers[i] = powers[j]; powers[j] = tp;
+            }
+        }
+    }
+
+    positions[0] = candidates[0];
+    if (positions_distinguish<N>(keys, positions.data(), 1, modulus)) { return 1; }
+
+    positions[0] = 0;
+    positions[1] = LAST_CHAR;
+    if (positions_distinguish<N>(keys, positions.data(), 2, modulus)) { return 2; }
+
+    {
+        std::size_t budget = 5000;
+        std::size_t num_found = 0;
+        if (backtracking_search<N>(keys, candidates.data(), num_candidates,
+                                   positions.data(), num_found, budget, modulus)) {
+            return num_found;
+        }
+    }
+
+    std::size_t num_pos = 0;
+    for (std::size_t ci = 0; ci < num_candidates && num_pos < MAX_POSITIONS; ++ci) {
+        bool already = false;
+        for (std::size_t p = 0; p < num_pos; ++p) {
+            if (positions[p] == candidates[ci]) { already = true; break; }
+        }
+        if (already) { continue; }
+        positions[num_pos] = candidates[ci];
+        ++num_pos;
+        if (positions_distinguish<N>(keys, positions.data(), num_pos, modulus)) { return num_pos; }
+    }
+
+    compile_time_error("Failed to find distinguishing positions for perfect hash");
+    return 0;
+}
+
+// Result of PHF computation. A max-sized slot_to_key array lets the same struct
+// type carry any chosen table size.
+template <std::size_t N>
+struct phf_result {
+    // Allow up to 8x the minimum table size. Sparser tables solve faster.
+    static constexpr std::size_t MAX_TABLE_SIZE = next_power_of_2(N) * 8;
+    std::size_t table_size{};
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso_values{};
+    std::size_t num_positions{};
+    std::array<std::size_t, MAX_POSITIONS> positions{};
+    std::array<std::size_t, MAX_TABLE_SIZE> slot_to_key{};
+};
+
+// Partition-based asso_values search (gperf-style). Determines asso_values one
+// (position, character) symbol at a time; never revisits a value. Equivalence
+// classes (keys sharing the same undetermined symbols) keep the search cheap.
+template <std::size_t N, std::size_t M>
+consteval bool try_generate_gperf(
+    const std::array<std::string_view, N>& keys,
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS>& asso_values,
+    std::size_t& num_positions,
+    std::array<std::size_t, MAX_POSITIONS>& positions,
+    std::array<std::size_t, M>& slot_to_key) {
+    num_positions = select_positions<N>(keys, positions, M);
+
+    for (std::size_t p = 0; p < MAX_POSITIONS; ++p) {
+        for (std::size_t c = 0; c < 256; ++c) { asso_values[p][c] = 0; }
+    }
+
+    if (num_positions == 0) {
+        for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+        for (std::size_t i = 0; i < N; ++i) {
+            std::size_t slot = keys[i].size() % M;
+            if (slot_to_key[slot] != N) { return false; }
+            slot_to_key[slot] = i;
+        }
+        return true;
+    }
+
+    std::array<std::array<std::size_t, MAX_POSITIONS>, N> kchars{};
+    for (std::size_t k = 0; k < N; ++k) {
+        for (std::size_t p = 0; p < num_positions; ++p) {
+            kchars[k][p] = char_at(keys[k], positions[p]);
+        }
+    }
+
+    struct sym_t { std::size_t pos; std::size_t ch; std::size_t freq; };
+    constexpr std::size_t MAX_SYMS = MAX_POSITIONS * 256;
+    std::array<sym_t, MAX_SYMS> syms{};
+    std::size_t nsyms = 0;
+    for (std::size_t p = 0; p < num_positions; ++p) {
+        std::array<std::size_t, 256> freq{};
+        for (std::size_t k = 0; k < N; ++k) {
+            std::size_t c = kchars[k][p];
+            if (c < 256) { freq[c]++; }
+        }
+        for (std::size_t c = 0; c < 256; ++c) {
+            if (freq[c] > 0) { syms[nsyms++] = {p, c, freq[c]}; }
+        }
+    }
+    for (std::size_t i = 0; i < nsyms; ++i) {
+        for (std::size_t j = i + 1; j < nsyms; ++j) {
+            if (syms[j].freq > syms[i].freq) {
+                auto tmp = syms[i]; syms[i] = syms[j]; syms[j] = tmp;
+            }
+        }
+    }
+
+    std::array<std::size_t, N> phash{};
+    for (std::size_t k = 0; k < N; ++k) { phash[k] = keys[k].size(); }
+
+    std::array<std::array<uint64_t, 256>, MAX_POSITIONS> salt{};
+    {
+        uint64_t s = 0x9e3779b97f4a7c15ULL;
+        for (std::size_t p = 0; p < num_positions; ++p) {
+            for (std::size_t c = 0; c < 256; ++c) {
+                s = s * 6364136223846793005ULL + 1442695040888963407ULL;
+                salt[p][c] = s;
+            }
+        }
+    }
+    std::array<uint64_t, N> sig{};
+    for (std::size_t k = 0; k < N; ++k) {
+        uint64_t s = 0;
+        for (std::size_t p = 0; p < num_positions; ++p) {
+            std::size_t c = kchars[k][p];
+            if (c < 256) { s ^= salt[p][c]; }
+        }
+        sig[k] = s;
+    }
+    std::array<std::size_t, N> order{};
+    for (std::size_t k = 0; k < N; ++k) { order[k] = k; }
+
+    std::array<std::size_t, M> slot_gen{};
+    std::size_t gen = 0;
+
+    std::size_t search_limit = next_power_of_2(M);
+    if (search_limit < 32) { search_limit = 32; }
+
+    for (std::size_t si = 0; si < nsyms; ++si) {
+        std::size_t sp = syms[si].pos;
+        std::size_t sc = syms[si].ch;
+
+        uint64_t sp_salt = salt[sp][sc];
+        for (std::size_t k = 0; k < N; ++k) {
+            if (kchars[k][sp] == sc) { sig[k] ^= sp_salt; }
+        }
+
+        for (std::size_t i = 1; i < N; ++i) {
+            std::size_t x = order[i];
+            uint64_t xs = sig[x];
+            std::size_t j = i;
+            while (j > 0 && sig[order[j - 1]] > xs) {
+                order[j] = order[j - 1];
+                --j;
+            }
+            order[j] = x;
+        }
+
+        bool found = false;
+        for (std::size_t v = 0; v < search_limit && !found; ++v) {
+            bool collision = false;
+            std::size_t ci = 0;
+            while (ci < N && !collision) {
+                uint64_t class_sig = sig[order[ci]];
+                std::size_t cj = ci;
+                while (cj < N && sig[order[cj]] == class_sig) { ++cj; }
+                if (cj - ci > 1) {
+                    ++gen;
+                    for (std::size_t x = ci; x < cj; ++x) {
+                        std::size_t k = order[x];
+                        std::size_t h = phash[k];
+                        if (kchars[k][sp] == sc) { h += v; }
+                        h %= M;
+                        if (slot_gen[h] == gen) { collision = true; break; }
+                        slot_gen[h] = gen;
+                    }
+                }
+                ci = cj;
+            }
+            if (!collision) {
+                asso_values[sp][sc] = v;
+                for (std::size_t k = 0; k < N; ++k) {
+                    if (kchars[k][sp] == sc) { phash[k] += v; }
+                }
+                found = true;
+            }
+        }
+        if (!found) { return false; }
+    }
+
+    for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+    for (std::size_t i = 0; i < N; ++i) {
+        std::size_t slot = phash[i] % M;
+        if (slot_to_key[slot] != N) { return false; }
+        slot_to_key[slot] = i;
+    }
+    std::size_t filled = 0;
+    for (std::size_t i = 0; i < M; ++i) {
+        if (slot_to_key[i] != N) { ++filled; }
+    }
+    return filled == N;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_compute_phf(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+    static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso{};
+    std::size_t npos{};
+    std::array<std::size_t, MAX_POSITIONS> pos{};
+    std::array<std::size_t, M> s2k{};
+    if (try_generate_gperf<N, M>(keys, asso, npos, pos, s2k)) {
+        result.table_size = M;
+        result.asso_values = asso;
+        result.num_positions = npos;
+        result.positions = pos;
+        for (std::size_t i = 0; i < M; ++i) { result.slot_to_key[i] = s2k[i]; }
+        for (std::size_t i = M; i < phf_result<N>::MAX_TABLE_SIZE; ++i) { result.slot_to_key[i] = N; }
+        return true;
+    }
+    return false;
+}
+
+template <std::size_t N, std::size_t M, std::size_t MaxM>
+consteval bool try_gperf_po2(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+    if (try_compute_phf<N, M>(keys, result)) { return true; }
+    constexpr std::size_t NextM = M * 2;
+    if constexpr (NextM <= MaxM) { return try_gperf_po2<N, NextM, MaxM>(keys, result); }
+    return false;
+}
+
+// --- Hash-and-Displace fallback --------------------------------------------
+
+constexpr std::size_t hd_bucket_hash(std::string_view key) noexcept {
+    std::size_t c0 = key.empty() ? 0 : static_cast<unsigned char>(key[0]);
+    std::size_t c1 = key.empty() ? 0 : static_cast<unsigned char>(key[key.size() - 1]);
+    return (c0 + c1 * 3 + key.size() * 17) & 0xFF;
+}
+constexpr std::size_t hd_safe_char(const char* p, std::size_t len, std::size_t idx) noexcept {
+    std::size_t has = static_cast<std::size_t>(idx < len);
+    std::size_t si = idx & (std::size_t{0} - has);
+    return static_cast<unsigned char>(p[si]) & (std::size_t{0} - has);
+}
+constexpr std::size_t hd_key_hash_2(std::string_view key) noexcept {
+    std::size_t kc = key.size();
+    kc = kc * 31 + static_cast<unsigned char>(key[0]);
+    kc = kc * 31 + hd_safe_char(key.data(), key.size(), 1);
+    return kc;
+}
+constexpr std::size_t hd_key_hash_4(std::string_view key) noexcept {
+    std::size_t kc = key.size();
+    kc = kc * 31 + static_cast<unsigned char>(key[0]);
+    kc = kc * 31 + hd_safe_char(key.data(), key.size(), 1);
+    kc = kc * 31 + hd_safe_char(key.data(), key.size(), 2);
+    kc = kc * 31 + hd_safe_char(key.data(), key.size(), 3);
+    return kc;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_hash_and_displace(
+    const std::array<std::string_view, N>& keys,
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS>& asso_values,
+    std::size_t& num_positions,
+    std::array<std::size_t, MAX_POSITIONS>& positions,
+    std::array<std::size_t, M>& slot_to_key) {
+    num_positions = select_positions<N>(keys, positions, M);
+
+    if (num_positions == 0) {
+        for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+        for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+        for (std::size_t i = 0; i < N; ++i) {
+            std::size_t slot = keys[i].size() % M;
+            if (slot_to_key[slot] != N) { return false; }
+            slot_to_key[slot] = i;
+        }
+        return true;
+    }
+
+    for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+    num_positions = HD_MODE; // sentinel for H&D mode
+    positions[0] = 0;
+    positions[1] = LAST_CHAR;
+
+    std::array<std::size_t, N> key_bucket{};
+    for (std::size_t i = 0; i < N; ++i) { key_bucket[i] = hd_bucket_hash(keys[i]); }
+
+    struct bucket_info { std::size_t ch; std::size_t count; };
+    std::array<bucket_info, N> buckets{};
+    std::size_t num_buckets = 0;
+    for (std::size_t i = 0; i < N; ++i) {
+        std::size_t bk = key_bucket[i];
+        bool found = false;
+        for (std::size_t b = 0; b < num_buckets; ++b) {
+            if (buckets[b].ch == bk) { ++buckets[b].count; found = true; break; }
+        }
+        if (!found) { buckets[num_buckets++] = {bk, 1}; }
+    }
+    for (std::size_t i = 0; i < num_buckets; ++i) {
+        for (std::size_t j = i + 1; j < num_buckets; ++j) {
+            if (buckets[j].count > buckets[i].count) {
+                auto tmp = buckets[i]; buckets[i] = buckets[j]; buckets[j] = tmp;
+            }
+        }
+    }
+
+    auto try_placement = [&](auto key_hash_fn) -> bool {
+        for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+        for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+        for (std::size_t b = 0; b < num_buckets; ++b) {
+            std::size_t ch = buckets[b].ch;
+            std::array<std::size_t, N> bucket_keys{};
+            std::size_t bk_count = 0;
+            for (std::size_t i = 0; i < N; ++i) {
+                if (key_bucket[i] == ch) { bucket_keys[bk_count++] = i; }
+            }
+            bool placed = false;
+            std::size_t max_d = M < 255 ? M : 255;
+            for (std::size_t d = 0; d < max_d; ++d) {
+                bool ok = true;
+                std::array<std::size_t, N> bucket_slots{};
+                for (std::size_t k = 0; k < bk_count; ++k) {
+                    std::size_t slot = (d + key_hash_fn(keys[bucket_keys[k]])) % M;
+                    if (slot_to_key[slot] != N) { ok = false; break; }
+                    for (std::size_t k2 = 0; k2 < k; ++k2) {
+                        if (bucket_slots[k2] == slot) { ok = false; break; }
+                    }
+                    if (!ok) { break; }
+                    bucket_slots[k] = slot;
+                }
+                if (ok) {
+                    asso_values[0][ch] = d;
+                    for (std::size_t k = 0; k < bk_count; ++k) {
+                        slot_to_key[bucket_slots[k]] = bucket_keys[k];
+                    }
+                    placed = true;
+                    break;
+                }
+            }
+            if (!placed) { return false; }
+        }
+        std::size_t filled = 0;
+        for (std::size_t i = 0; i < M; ++i) {
+            if (slot_to_key[i] != N) { ++filled; }
+        }
+        return filled == N;
+    };
+
+    if (try_placement([](std::string_view k) { return hd_key_hash_2(k); })) {
+        positions[2] = HD_HASH_2BYTE_FLAG;
+        return true;
+    }
+    if (try_placement([](std::string_view k) { return hd_key_hash_4(k); })) {
+        positions[2] = HD_HASH_4BYTE_FLAG;
+        return true;
+    }
+    return false;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_compute_phf_hd(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+    static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso{};
+    std::size_t npos{};
+    std::array<std::size_t, MAX_POSITIONS> pos{};
+    std::array<std::size_t, M> s2k{};
+    if (try_hash_and_displace<N, M>(keys, asso, npos, pos, s2k)) {
+        result.table_size = M;
+        result.asso_values = asso;
+        result.num_positions = npos;
+        result.positions = pos;
+        for (std::size_t i = 0; i < M; ++i) { result.slot_to_key[i] = s2k[i]; }
+        for (std::size_t i = M; i < phf_result<N>::MAX_TABLE_SIZE; ++i) { result.slot_to_key[i] = N; }
+        return true;
+    }
+    return false;
+}
+
+template <std::size_t N, std::size_t M>
+consteval phf_result<N> compute_phf_hd_po2(const std::array<std::string_view, N>& keys) {
+    static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+    phf_result<N> result{};
+    if (try_compute_phf_hd<N, M>(keys, result)) { return result; }
+    constexpr std::size_t NextM = M * 2;
+    if constexpr (NextM <= phf_result<N>::MAX_TABLE_SIZE) {
+        return compute_phf_hd_po2<N, NextM>(keys);
+    } else {
+        compile_time_error("Hash-and-Displace: failed to find valid table size");
+        return result;
+    }
+}
+
+// Compute a perfect hash for `keys`: try gperf at power-of-two sizes (capped so
+// the runtime tables stay within uint8 indices), then fall back to H&D.
+template <std::size_t N>
+consteval phf_result<N> compute_phf(const std::array<std::string_view, N>& keys) {
+    constexpr std::size_t StartM = next_power_of_2(N);
+    constexpr std::size_t GPERF_MAX_TABLE =
+        phf_result<N>::MAX_TABLE_SIZE < 256 ? phf_result<N>::MAX_TABLE_SIZE : 256;
+    if constexpr (StartM <= GPERF_MAX_TABLE) {
+        phf_result<N> result{};
+        if (try_gperf_po2<N, StartM, GPERF_MAX_TABLE>(keys, result)) { return result; }
+    }
+    return compute_phf_hd_po2<N, StartM>(keys);
+}
+
+// ============================================================================
+// Runtime tables (flat, uint8) derived from a phf_result.
+// ============================================================================
+
+template <std::size_t N, std::size_t TableSize, std::size_t MaxKeyLen>
+struct phf_data {
+    std::array<std::array<std::uint8_t, 256>, MAX_POSITIONS> asso_values{};
+    std::array<std::uint8_t, MAX_POSITIONS>                  positions{};
+    std::uint8_t                                             num_positions{};
+    std::uint8_t                                             hd_hash_variant{}; // 2 or 4 (H&D only)
+    std::array<std::uint8_t, TableSize>                      slot_to_key{};
+    // slot_key_bytes[s] holds the key stored at slot s, zero-padded to a 16-byte
+    // multiple so the SIMD comparison can read a whole register.
+    std::array<std::array<char, ((MaxKeyLen + 15) / 16) * 16>, TableSize> slot_key_bytes{};
+    std::array<std::uint8_t, TableSize>                      slot_key_len{};
+};
+
+template <std::size_t N>
+constexpr std::size_t compute_max_key_len(const std::array<std::string_view, N>& keys) noexcept {
+    std::size_t m = 0;
+    for (std::size_t i = 0; i < N; ++i) { if (keys[i].size() > m) { m = keys[i].size(); } }
+    return m;
+}
+
+// Validate keys and build the runtime tables from the computed perfect hash.
+template <std::size_t N, std::size_t TableSize, std::size_t MaxKeyLen>
+consteval phf_data<N, TableSize, MaxKeyLen>
+build_phf_data(const std::array<std::string_view, N>& keys, const phf_result<N>& result) {
+    for (std::size_t i = 0; i < N; ++i) {
+        if (keys[i].empty())            { compile_time_error("empty keys are not allowed in key_selector"); }
+        if (keys[i].size() > MaxKeyLen) { compile_time_error("key length exceeds MaxKeyLen"); }
+        for (char c : keys[i]) {
+            if (c == '\\') { compile_time_error("backslash not allowed in key_selector keys"); }
+            if (c == '"')  { compile_time_error("quote not allowed in key_selector keys"); }
+            if (c == '\0') { compile_time_error("null byte not allowed in key_selector keys"); }
+        }
+        for (std::size_t j = i + 1; j < N; ++j) {
+            if (keys[i] == keys[j]) { compile_time_error("duplicate keys in key_selector"); }
+        }
+    }
+
+    phf_data<N, TableSize, MaxKeyLen> out{};
+
+    if (result.num_positions == HD_MODE) {
+        // H&D mode: single displacement table in asso_values[0].
+        for (std::size_t c = 0; c < 256; ++c) {
+            out.asso_values[0][c] = static_cast<std::uint8_t>(result.asso_values[0][c]);
+        }
+        out.num_positions   = static_cast<std::uint8_t>(HD_MODE);
+        out.hd_hash_variant = static_cast<std::uint8_t>(result.positions[2]);
+    } else {
+        for (std::size_t pi = 0; pi < result.num_positions; ++pi) {
+            for (std::size_t c = 0; c < 256; ++c) {
+                out.asso_values[pi][c] = static_cast<std::uint8_t>(result.asso_values[pi][c] % TableSize);
+            }
+        }
+        out.num_positions = static_cast<std::uint8_t>(result.num_positions);
+        for (std::size_t i = 0; i < result.num_positions; ++i) {
+            out.positions[i] = (result.positions[i] == LAST_CHAR)
+                ? POS_LAST_CHAR
+                : static_cast<std::uint8_t>(result.positions[i]);
+        }
+    }
+
+    for (std::size_t s = 0; s < TableSize; ++s) {
+        out.slot_to_key[s] = static_cast<std::uint8_t>(result.slot_to_key[s]);
+    }
+
+    for (std::size_t s = 0; s < TableSize; ++s) {
+        std::size_t ki = result.slot_to_key[s];
+        if (ki < N) {
+            auto k = keys[ki];
+            out.slot_key_len[s] = static_cast<std::uint8_t>(k.size());
+            for (std::size_t c = 0; c < k.size(); ++c) { out.slot_key_bytes[s][c] = k[c]; }
+        } else {
+            out.slot_key_len[s] = 0; // empty slot: no length can match
+        }
+    }
+    return out;
+}
+
+// --- SIMD runtime primitives ------------------------------------------------
+
+// The runtime matchers read whole 16-byte blocks up to offset 63 (four blocks)
+// for keys as long as the 63-character maximum. That read must stay within the
+// buffer's trailing padding, so the padding has to exceed the largest offset we
+// touch.
+static_assert(SIMDJSON_PADDING > 63,
+              "key_selector requires SIMDJSON_PADDING > 63 for its SIMD key reads");
+
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+// True when every byte of `diff` is zero. A 32-bit-lane horizontal max is
+// enough (zero iff every 32-bit word is zero) and is cheaper than a byte-wide
+// reduction. Do not replace this with a floating-point compare against 0.0:
+// flush-to-zero would treat a denormal as zero.
+simdjson_really_inline bool neon_all_bytes_zero(uint8x16_t diff) noexcept {
+    return vmaxvq_u32(vreinterpretq_u32_u8(diff)) == 0;
+}
+#endif
+
+// Scan for the terminating '"' starting at p. Returns its byte offset (= key
+// length). Caller guarantees SIMDJSON_PADDING (== 64) bytes past the JSON buffer,
+// so reading whole 16-byte blocks up to offset 63 is always safe.
+template <std::size_t MaxKeyLen>
+simdjson_really_inline std::size_t scan_key_length(const char* p) noexcept {
+    // The closing quote of the longest key sits at most at offset MaxKeyLen, so we
+    // scan as many 16-byte blocks as it takes to cover offsets 0..MaxKeyLen. The
+    // cap (63) keeps every covered offset within the 64-byte padding guarantee, so
+    // the SIMD and scalar builds agree.
+    static_assert(MaxKeyLen <= 63, "MaxKeyLen must be <= 63 for current SIMD implementations");
+    // [[maybe_unused]]: the scalar fallback (no NEON/SSE2/LSX) does not use it.
+    [[maybe_unused]] constexpr std::size_t num_blocks = MaxKeyLen / 16 + 1;
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+    for (std::size_t b = 0; b < num_blocks; ++b) {
+        uint8x16_t v = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + b * 16);
+        uint8x16_t cmp = vceqq_u8(v, vdupq_n_u8('"'));
+        uint64_t m = vget_lane_u64(
+            vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(cmp), 4)), 0);
+        if (simdjson_likely(m != 0)) { return b * 16 + (std::size_t(__builtin_ctzll(m)) >> 2); }
+    }
+    return MaxKeyLen + 1;
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+    for (std::size_t b = 0; b < num_blocks; ++b) {
+        __m128i v   = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + b * 16));
+        __m128i cmp = _mm_cmpeq_epi8(v, _mm_set1_epi8('"'));
+        unsigned m  = static_cast<unsigned>(_mm_movemask_epi8(cmp));
+        if (simdjson_likely(m != 0)) { return b * 16 + std::size_t(__builtin_ctz(m)); }
+    }
+    return MaxKeyLen + 1;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+    for (std::size_t b = 0; b < num_blocks; ++b) {
+        __m128i v   = __lsx_vld(reinterpret_cast<const void*>(p + b * 16), 0);
+        __m128i cmp = __lsx_vseq_b(v, __lsx_vreplgr2vr_b('"'));
+        // vmskltz_b gathers the per-byte sign bits (set where the byte equals '"')
+        // into the low 16 bits of lane 0, the LSX equivalent of movemask.
+        unsigned m  = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(cmp), 0)) & 0xFFFFu;
+        if (simdjson_likely(m != 0)) { return b * 16 + std::size_t(__builtin_ctz(m)); }
+    }
+    return MaxKeyLen + 1;
+#else
+    for (std::size_t i = 0; i <= MaxKeyLen; ++i)
+        if (p[i] == '"') return i;
+    return MaxKeyLen + 1;
+#endif
+}
+
+// Byte-equal of p[0..len) against stored[0..len). stored is zero-padded past
+// `len`. Input is read over 16, 32, 48, or 64 bytes (padded JSON buffer
+// guaranteed; SIMDJSON_PADDING == 64).
+template <std::size_t MaxKeyLen>
+simdjson_really_inline bool compare_key_bytes(
+    const char* p, const char* stored, std::size_t len) noexcept {
+    // [[maybe_unused]]: only the NEON/SSE2/LSX branches read these; the scalar
+    // fallback build (no SIMD) leaves them unused, which is an error under -Werror.
+    [[maybe_unused]] alignas(16) static constexpr uint8_t idx16[16] =
+        {0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15};
+    if constexpr (MaxKeyLen <= 16) {
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+        uint8x16_t vp = vld1q_u8(reinterpret_cast<const uint8_t*>(p));
+        uint8x16_t vs = vld1q_u8(reinterpret_cast<const uint8_t*>(stored));
+        uint8x16_t mask = vcltq_u8(vld1q_u8(idx16), vdupq_n_u8(static_cast<uint8_t>(len)));
+        uint8x16_t diff = veorq_u8(vandq_u8(vp, mask), vs);
+        return neon_all_bytes_zero(diff);
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+        __m128i vp = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p));
+        __m128i vs = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored));
+        __m128i idx = _mm_load_si128(reinterpret_cast<const __m128i*>(idx16));
+        __m128i mask = _mm_cmplt_epi8(idx, _mm_set1_epi8(static_cast<char>(len)));
+        __m128i eq = _mm_cmpeq_epi8(_mm_and_si128(vp, mask), vs);
+        return _mm_movemask_epi8(eq) == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+        __m128i vp = __lsx_vld(reinterpret_cast<const void*>(p), 0);
+        __m128i vs = __lsx_vld(reinterpret_cast<const void*>(stored), 0);
+        __m128i idx = __lsx_vld(reinterpret_cast<const void*>(idx16), 0);
+        __m128i mask = __lsx_vslt_b(idx, __lsx_vreplgr2vr_b(static_cast<int>(len)));
+        __m128i eq = __lsx_vseq_b(__lsx_vand_v(vp, mask), vs);
+        return (static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq), 0)) & 0xFFFFu) == 0xFFFFu;
+#else
+        for (std::size_t i = 0; i < len; ++i)
+            if (p[i] != stored[i]) return false;
+        return true;
+#endif
+    } else if constexpr (MaxKeyLen <= 32) {
+        [[maybe_unused]] alignas(16) static constexpr uint8_t idx32_hi[16] =
+            {16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31};
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+        uint8x16_t vp_lo = vld1q_u8(reinterpret_cast<const uint8_t*>(p));
+        uint8x16_t vp_hi = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + 16);
+        uint8x16_t vs_lo = vld1q_u8(reinterpret_cast<const uint8_t*>(stored));
+        uint8x16_t vs_hi = vld1q_u8(reinterpret_cast<const uint8_t*>(stored) + 16);
+        uint8x16_t lenv = vdupq_n_u8(static_cast<uint8_t>(len));
+        uint8x16_t m_lo = vcltq_u8(vld1q_u8(idx16),    lenv);
+        uint8x16_t m_hi = vcltq_u8(vld1q_u8(idx32_hi), lenv);
+        uint8x16_t d_lo = veorq_u8(vandq_u8(vp_lo, m_lo), vs_lo);
+        uint8x16_t d_hi = veorq_u8(vandq_u8(vp_hi, m_hi), vs_hi);
+        return neon_all_bytes_zero(vorrq_u8(d_lo, d_hi));
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+        __m128i vp_lo = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p));
+        __m128i vp_hi = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + 16));
+        __m128i vs_lo = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored));
+        __m128i vs_hi = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored + 16));
+        __m128i lenv = _mm_set1_epi8(static_cast<char>(len));
+        __m128i m_lo = _mm_cmplt_epi8(_mm_load_si128(reinterpret_cast<const __m128i*>(idx16)),    lenv);
+        __m128i m_hi = _mm_cmplt_epi8(_mm_load_si128(reinterpret_cast<const __m128i*>(idx32_hi)), lenv);
+        __m128i eq_lo = _mm_cmpeq_epi8(_mm_and_si128(vp_lo, m_lo), vs_lo);
+        __m128i eq_hi = _mm_cmpeq_epi8(_mm_and_si128(vp_hi, m_hi), vs_hi);
+        return (_mm_movemask_epi8(eq_lo) & _mm_movemask_epi8(eq_hi)) == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+        __m128i lenv = __lsx_vreplgr2vr_b(static_cast<int>(len));
+        __m128i vp_lo = __lsx_vld(reinterpret_cast<const void*>(p), 0);
+        __m128i vp_hi = __lsx_vld(reinterpret_cast<const void*>(p + 16), 0);
+        __m128i vs_lo = __lsx_vld(reinterpret_cast<const void*>(stored), 0);
+        __m128i vs_hi = __lsx_vld(reinterpret_cast<const void*>(stored + 16), 0);
+        __m128i m_lo = __lsx_vslt_b(__lsx_vld(reinterpret_cast<const void*>(idx16), 0),    lenv);
+        __m128i m_hi = __lsx_vslt_b(__lsx_vld(reinterpret_cast<const void*>(idx32_hi), 0), lenv);
+        __m128i eq_lo = __lsx_vseq_b(__lsx_vand_v(vp_lo, m_lo), vs_lo);
+        __m128i eq_hi = __lsx_vseq_b(__lsx_vand_v(vp_hi, m_hi), vs_hi);
+        unsigned mlo = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq_lo), 0)) & 0xFFFFu;
+        unsigned mhi = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq_hi), 0)) & 0xFFFFu;
+        return (mlo & mhi) == 0xFFFFu;
+#else
+        for (std::size_t i = 0; i < len; ++i)
+            if (p[i] != stored[i]) return false;
+        return true;
+#endif
+    } else if constexpr (MaxKeyLen <= 64) {
+        // 3 or 4 16-byte blocks (33..48 -> 3, 49..64 -> 4). stored is zero-padded
+        // to exactly num_blocks*16 bytes (KEY_STRIDE), so neither load overruns it.
+        // [[maybe_unused]]: the scalar fallback (no NEON/SSE2/LSX) does not use it.
+        [[maybe_unused]] constexpr std::size_t num_blocks = (MaxKeyLen + 15) / 16;
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+        uint8x16_t base = vld1q_u8(idx16);
+        uint8x16_t lenv = vdupq_n_u8(static_cast<uint8_t>(len));
+        uint8x16_t acc  = vdupq_n_u8(0);
+        for (std::size_t b = 0; b < num_blocks; ++b) {
+            uint8x16_t vp   = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + b * 16);
+            uint8x16_t vs   = vld1q_u8(reinterpret_cast<const uint8_t*>(stored) + b * 16);
+            uint8x16_t idxv = vaddq_u8(base, vdupq_n_u8(static_cast<uint8_t>(b * 16)));
+            uint8x16_t mask = vcltq_u8(idxv, lenv);
+            acc = vorrq_u8(acc, veorq_u8(vandq_u8(vp, mask), vs));
+        }
+        return neon_all_bytes_zero(acc);
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+        __m128i base = _mm_load_si128(reinterpret_cast<const __m128i*>(idx16));
+        __m128i lenv = _mm_set1_epi8(static_cast<char>(len));
+        int eq = 0xFFFF;
+        for (std::size_t b = 0; b < num_blocks; ++b) {
+            __m128i vp   = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + b * 16));
+            __m128i vs   = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored + b * 16));
+            __m128i idxv = _mm_add_epi8(base, _mm_set1_epi8(static_cast<char>(b * 16)));
+            __m128i mask = _mm_cmplt_epi8(idxv, lenv);
+            eq &= _mm_movemask_epi8(_mm_cmpeq_epi8(_mm_and_si128(vp, mask), vs));
+        }
+        return eq == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+        __m128i base = __lsx_vld(reinterpret_cast<const void*>(idx16), 0);
+        __m128i lenv = __lsx_vreplgr2vr_b(static_cast<int>(len));
+        unsigned acc = 0xFFFFu;
+        for (std::size_t b = 0; b < num_blocks; ++b) {
+            __m128i vp   = __lsx_vld(reinterpret_cast<const void*>(p + b * 16), 0);
+            __m128i vs   = __lsx_vld(reinterpret_cast<const void*>(stored + b * 16), 0);
+            __m128i idxv = __lsx_vadd_b(base, __lsx_vreplgr2vr_b(static_cast<int>(b * 16)));
+            __m128i mask = __lsx_vslt_b(idxv, lenv);
+            __m128i eq   = __lsx_vseq_b(__lsx_vand_v(vp, mask), vs);
+            acc &= static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq), 0)) & 0xFFFFu;
+        }
+        return acc == 0xFFFFu;
+#else
+        for (std::size_t i = 0; i < len; ++i)
+            if (p[i] != stored[i]) return false;
+        return true;
+#endif
+    } else {
+        for (std::size_t i = 0; i < len; ++i)
+            if (p[i] != stored[i]) return false;
+        return true;
+    }
+}
+
+// --- Single 8-bit window fast path ------------------------------------------
+//
+// Many small key sets can be told apart by inspecting a *single* 8-bit window of
+// the key bytes -- and that window need not be byte-aligned. Because every JSON
+// key is terminated by a '"', the bytes at and before a key's length are well
+// defined for any key at least that long: byte i is the key character when i is
+// inside the key and the closing quote when i == len. So we read two bytes at a
+// fixed offset, extract 8 consecutive bits at a fixed intra-byte shift, and if
+// that value is distinct for every key, a 256-entry table maps it straight to a
+// candidate key. The match then needs no hash and -- in the length-free overload
+// -- no SIMD length scan: load two bytes, shift, mask, index the table, and
+// confirm the candidate with one comparison.
+//
+// Allowing an *unaligned* window (a shift of 1..7) mixes bits from two adjacent
+// bytes and discriminates key sets that no single aligned byte can. For example,
+// the partial_tweets keys {created_at,id,text,in_reply_to_status_id,user,
+// retweet_count,favorite_count} share a colliding byte at every aligned position
+// 0,1,2, yet the 8 bits starting at bit offset 2 are unique across all seven.
+// Simpler cases fall out as the shift==0 special case: {"id","screen_name"}
+// splits on byte 0, {"jo","joe"} on the quote at byte 2.
+//
+// The window is confined to the first (shortest key length + 1) bytes so the
+// two-byte read never crosses a key's closing quote into uncontrolled value
+// bytes; that final byte is the shortest key's quote.
+template <std::size_t N, std::size_t MaxKeyLen>
+struct window_data {
+    static constexpr std::size_t KEY_STRIDE = ((MaxKeyLen + 15) / 16) * 16;
+    bool                                            ok{false};
+    std::uint8_t                                    byte_offset{0}; // first byte of the 2-byte read
+    std::uint8_t                                    shift{0};       // intra-byte bit shift (0..7)
+    std::array<std::uint8_t, 256>                   window_to_key{}; // window byte -> key index, N if none
+    std::array<std::uint8_t, N>                     key_len{};
+    std::array<std::array<char, KEY_STRIDE>, N>     key_bytes{};     // zero-padded
+};
+
+// Byte seen at `idx` for key `i`: a key character when idx is inside the key, or
+// the closing '"' when idx == len (callers keep idx <= every key's length).
+template <std::size_t N>
+constexpr unsigned window_byte_at(const std::array<std::string_view, N>& keys,
+                                  std::size_t i, std::size_t idx) noexcept {
+    if (idx < keys[i].size()) { return static_cast<unsigned char>(keys[i][idx]); }
+    return static_cast<unsigned>('"');
+}
+
+// The 8-bit window value for key `i` at (byte_offset, shift): the little-endian
+// pair (byte[off], byte[off+1]) shifted right by `shift` and truncated. Matches
+// the runtime read exactly.
+template <std::size_t N>
+constexpr unsigned window_value(const std::array<std::string_view, N>& keys,
+                                std::size_t i, std::size_t byte_offset, std::size_t shift) noexcept {
+    unsigned lo = window_byte_at<N>(keys, i, byte_offset);
+    unsigned hi = window_byte_at<N>(keys, i, byte_offset + 1);
+    return ((lo | (hi << 8)) >> shift) & 0xFFu;
+}
+
+template <std::size_t N, std::size_t MaxKeyLen>
+consteval window_data<N, MaxKeyLen>
+compute_window(const std::array<std::string_view, N>& keys) {
+    window_data<N, MaxKeyLen> out{};
+
+    std::size_t min_len = keys[0].size();
+    for (std::size_t i = 1; i < N; ++i) {
+        if (keys[i].size() < min_len) { min_len = keys[i].size(); }
+    }
+
+    // Iterate windows nearest the front first (cheapest to read, smallest shift).
+    for (std::size_t off = 0; off <= min_len; ++off) {
+        for (std::size_t shift = 0; shift < 8; ++shift) {
+            // The read touches byte off, and byte off+1 when shift != 0. Both must
+            // stay within the safe region [0, min_len] (min_len is the shortest
+            // key's quote index). off <= min_len is guaranteed by the loop bound.
+            if (shift != 0 && off + 1 > min_len) { continue; }
+
+            bool distinct = true;
+            for (std::size_t i = 0; i < N && distinct; ++i) {
+                for (std::size_t j = i + 1; j < N; ++j) {
+                    if (window_value<N>(keys, i, off, shift) == window_value<N>(keys, j, off, shift)) {
+                        distinct = false;
+                        break;
+                    }
+                }
+            }
+            if (!distinct) { continue; }
+
+            out.ok          = true;
+            out.byte_offset = static_cast<std::uint8_t>(off);
+            out.shift       = static_cast<std::uint8_t>(shift);
+            for (std::size_t b = 0; b < 256; ++b) { out.window_to_key[b] = static_cast<std::uint8_t>(N); }
+            for (std::size_t i = 0; i < N; ++i) {
+                out.window_to_key[window_value<N>(keys, i, off, shift)] = static_cast<std::uint8_t>(i);
+                out.key_len[i] = static_cast<std::uint8_t>(keys[i].size());
+                for (std::size_t c = 0; c < keys[i].size(); ++c) { out.key_bytes[i][c] = keys[i][c]; }
+            }
+            return out;
+        }
+    }
+    return out; // ok == false: no single 8-bit window distinguishes the keys
+}
+
+// Read the 8-bit window at (byte_offset, shift) from a padded key pointer.
+simdjson_really_inline std::uint8_t read_window(const char* p, std::size_t byte_offset,
+                                                std::size_t shift) noexcept {
+    std::uint16_t w;
+    // Two controlled bytes (within the shortest key + its quote, hence within the
+    // padded buffer). memcpy is the portable little-endian unaligned load.
+    std::memcpy(&w, p + byte_offset, sizeof(w));
+#if defined(__BYTE_ORDER__) && (__BYTE_ORDER__ == __ORDER_BIG_ENDIAN__)
+    w = static_cast<std::uint16_t>((w >> 8) | (w << 8));
+#endif
+    return static_cast<std::uint8_t>((w >> shift) & 0xFFu);
+}
+
+// Verify a window candidate. The window table already mapped the key to the
+// single possible index `ki` (< N); here we confirm it. Folding over 0..N-1
+// turns the runtime `ki` into a compile-time index in the matching arm, so the
+// candidate's length and bytes are constants for compare_key_bytes -- the same
+// specialization the ordered find_field path gets from a CT-length
+// unsafe_is_equal, and what keeps this path competitive. The closing-quote check
+// (p[L] == '"') both confirms the key ends exactly at the candidate's length and
+// rejects a wrong-length key, so this works whether or not the caller knew len.
+template <std::size_t N, std::size_t MaxKeyLen, std::size_t... Is>
+simdjson_really_inline std::size_t
+match_window_candidate(const char* p, std::uint8_t ki,
+                       const window_data<N, MaxKeyLen>& w,
+                       std::index_sequence<Is...>) noexcept {
+  std::size_t result = N;
+  auto try_match = [&](auto Ic) {
+    constexpr std::size_t i = decltype(Ic)::value;
+    if (ki == i && p[w.key_len[i]] == '"' &&
+        key_selector_detail::compare_key_bytes<MaxKeyLen>(
+            p, w.key_bytes[i].data(), w.key_len[i])) {
+      result = i;
+    }
+  };
+  (try_match(std::integral_constant<std::size_t, Is>{}), ...);
+  return result;
+}
+
+// --- describe() string helpers (constexpr; no std::to_string, which is not) ---
+
+// Append the decimal form of v to s.
+SIMDJSON_CONSTEXPR_STRING void append_uint(std::string& s, std::size_t v) {
+    if (v == 0) { s.push_back('0'); return; }
+    char buf[20];
+    std::size_t n = 0;
+    while (v > 0) { buf[n++] = static_cast<char>('0' + (v % 10)); v /= 10; }
+    while (n > 0) { s.push_back(buf[--n]); }
+}
+
+// Append a byte as its decimal value, plus the printable character in quotes
+// when it is in the printable ASCII range (e.g. "110 ('n')").
+SIMDJSON_CONSTEXPR_STRING void append_byte(std::string& s, unsigned b) {
+    append_uint(s, b);
+    if (b >= 0x20 && b < 0x7f) {
+        s += " ('";
+        s.push_back(static_cast<char>(b));
+        s += "')";
+    }
+}
+
+} // namespace key_selector_detail
+
+/**
+ * Stateless, compile-time key selector.
+ *
+ * Usage:
+ *   using sel_t = key_selector<"id", "text", "user">;
+ *   std::size_t i = sel_t::match_raw(raw_key); // returns sel_t::size() on miss
+ *
+ * The perfect hash is built at compile time (gperf-style, with a
+ * Hash-and-Displace fallback) and only flat tables survive to runtime. All
+ * tables are static constexpr, so the lookup fully inlines.
+ *
+ * Limitations:
+ *   - Each key must be at most 63 characters long (and no longer than
+ *     SIMDJSON_PADDING). Longer keys trigger a compile-time error.
+ *   - The number of keys should be moderate. The hard limit is 255 keys;
+ *     compilation time grows with the number of keys, so prefer a few dozen at
+ *     most per selector.
+ *   - Keys must be distinct, non-empty, and free of backslash, double-quote and
+ *     null bytes (matching is done against the raw, unescaped JSON key bytes).
+ */
+template <constevalutil::fixed_string... Keys>
+struct key_selector {
+    static constexpr std::size_t N = sizeof...(Keys);
+    static_assert(N > 0,   "key_selector requires at least one key");
+    static_assert(N <= 255,"key_selector supports at most 255 keys");
+
+    static constexpr std::array<std::string_view, N> keys{ Keys.view()... };
+    static constexpr std::size_t max_key_len = key_selector_detail::compute_max_key_len<N>(keys);
+    static_assert(max_key_len <= SIMDJSON_PADDING,
+                  "key longer than SIMDJSON_PADDING is not supported");
+    // The SIMD key-length scan covers offsets 0..63 (four 16-byte blocks), which
+    // stays within the 64-byte padding guarantee. A 64-character key's closing
+    // quote would land at offset 64 and be missed on NEON/SSE2/LSX while still
+    // matching in scalar builds, so cap at 63 to keep implementations in agreement.
+    static_assert(max_key_len <= 63,
+                  "key_selector keys must be at most 63 characters long");
+
+    static constexpr auto result = key_selector_detail::compute_phf<N>(keys);
+    static constexpr std::size_t table_size = result.table_size;
+
+    static constexpr auto phf =
+        key_selector_detail::build_phf_data<N, table_size, max_key_len>(keys, result);
+
+    // Single 8-bit-window discriminator (when one exists). Detected at compile
+    // time and selected with `if constexpr` below, so the hash path is compiled
+    // out for key sets that qualify, and this is compiled out for those that do
+    // not.
+    static constexpr auto window =
+        key_selector_detail::compute_window<N, max_key_len>(keys);
+
+    static constexpr std::size_t size() noexcept { return N; }
+
+    /**
+     * Look up a JSON key whose length is already known. p must point at the first
+     * key byte (just after the opening quote) in a padded simdjson buffer, and len
+     * must be the number of raw key bytes (the distance to the closing quote).
+     * Returns the selector index in [0, N) on match, or N on miss.
+     *
+     * Prefer this overload when the caller can obtain the key length cheaply (for
+     * example, object::for_each derives it from the structural index rather than
+     * re-scanning for the closing quote).
+     */
+    static simdjson_really_inline std::size_t match_raw(const char* p, std::size_t len) noexcept {
+        if (len == 0 || len > max_key_len) { return N; }
+
+        if constexpr (window.ok) {
+            // One 8-bit window selects the only possible candidate key;
+            // match_window_candidate confirms it (bytes + closing quote). p sits
+            // in a padded buffer and the window stays within the shortest key +
+            // quote, so the two-byte read is always in bounds. len is unused here
+            // because the quote check already pins the key's end.
+            std::uint8_t ki = window.window_to_key[
+                key_selector_detail::read_window(p, window.byte_offset, window.shift)];
+            if (ki >= N) { return N; }
+            return key_selector_detail::match_window_candidate(
+                p, ki, window, std::make_index_sequence<N>{});
+        }
+
+        std::size_t slot;
+        if (phf.num_positions == key_selector_detail::HD_MODE) {
+            // Hash-and-Displace: bucket displacement + per-key hash.
+            std::string_view key(p, len);
+            std::size_t bucket = key_selector_detail::hd_bucket_hash(key);
+            std::size_t kh = (phf.hd_hash_variant == key_selector_detail::HD_HASH_2BYTE_FLAG)
+                ? key_selector_detail::hd_key_hash_2(key)
+                : key_selector_detail::hd_key_hash_4(key);
+            slot = (phf.asso_values[0][bucket] + kh) & (table_size - 1);
+        } else {
+            // gperf: h = len + sum of asso_values over the selected positions.
+            // positions / num_positions / asso_values are compile-time constants,
+            // so this loop fully unrolls. The idx < len guard mirrors the
+            // generator's char_at()-> 256 -> skip behavior for out-of-range
+            // positions (required: arbitrary positions may exceed a key's length).
+            std::size_t h = len;
+            for (std::uint8_t i = 0; i < phf.num_positions; ++i) {
+                std::uint8_t pos = phf.positions[i];
+                std::size_t idx = (pos == key_selector_detail::POS_LAST_CHAR)
+                                  ? (len - std::size_t{1})
+                                  : static_cast<std::size_t>(pos);
+                if (idx < len) {
+                    h += phf.asso_values[i][static_cast<unsigned char>(p[idx])];
+                }
+            }
+            slot = h & (table_size - 1);
+        }
+
+        std::uint8_t ki = phf.slot_to_key[slot];
+        if (ki >= N) { return N; }
+        if (phf.slot_key_len[slot] != len) { return N; }
+        if (!key_selector_detail::compare_key_bytes<max_key_len>(
+                p, phf.slot_key_bytes[slot].data(), len)) { return N; }
+        return ki;
+    }
+
+    /**
+     * Look up a JSON key. rjs must point just after an opening quote in a padded
+     * simdjson buffer. Returns the selector index in [0, N) on match, or N on miss.
+     * The key length is recovered with a SIMD scan for the closing quote; callers
+     * that already know the length should use the (p, len) overload above.
+     */
+    static simdjson_really_inline std::size_t match_raw(raw_json_string rjs) noexcept {
+        const char* p = rjs.raw();
+        if constexpr (window.ok) {
+            // One 8-bit window picks the candidate; verifying the candidate's
+            // bytes and its closing '"' confirms the full key, so the length scan
+            // is unnecessary. The window read is in bounds (padding), and the
+            // candidate length is at most max_key_len.
+            std::uint8_t ki = window.window_to_key[
+                key_selector_detail::read_window(p, window.byte_offset, window.shift)];
+            if (ki >= N) { return N; }
+            return key_selector_detail::match_window_candidate(
+                p, ki, window, std::make_index_sequence<N>{});
+        }
+        return match_raw(p, key_selector_detail::scan_key_length<max_key_len>(p));
+    }
+
+    /** Return the key text at selector index i (i in [0, N)). */
+    static constexpr std::string_view key_at(std::size_t i) noexcept {
+        return keys[i];
+    }
+
+    /**
+     * Return a complete, human-readable, multi-line description of how this
+     * selector classifies a key: which algorithm was selected at compile time
+     * (single 8-bit window, gperf-style perfect hash, or hash-and-displace), the
+     * exact bytes/positions it inspects, and the contents of the lookup tables
+     * (which window bytes or hash slots map to which key). The text mirrors what
+     * match_raw() does step by step.
+     *
+     * Everything it reports is derived from the compile-time tables, so describe()
+     * is itself usable in a constant expression when the standard library supports
+     * constexpr std::string (__cpp_lib_constexpr_string):
+     *
+     *   static_assert(!key_selector<"name", "city">::describe().empty());
+     *
+     * It allocates a std::string and is meant for documentation, debugging and
+     * tests, not for any hot path.
+     */
+    static SIMDJSON_CONSTEXPR_STRING std::string describe() {
+        std::string s;
+        s += "key_selector: ";
+        key_selector_detail::append_uint(s, N);
+        s += " keys, max key length ";
+        key_selector_detail::append_uint(s, max_key_len);
+        s += "\nkeys:\n";
+        for (std::size_t i = 0; i < N; ++i) {
+            s += "  [";
+            key_selector_detail::append_uint(s, i);
+            s += "] \"";
+            s += keys[i];
+            s += "\" (length ";
+            key_selector_detail::append_uint(s, keys[i].size());
+            s += ")\n";
+        }
+        if constexpr (window.ok) {
+            // Mirrors the window fast path of match_raw().
+            s += "algorithm: single 8-bit window\n";
+            s += "  step 1: read 2 bytes at offset ";
+            key_selector_detail::append_uint(s, window.byte_offset);
+            s += ", interpret them as a little-endian 16-bit value, shift right by ";
+            key_selector_detail::append_uint(s, window.shift);
+            s += " bits, and keep the low 8 bits\n";
+            s += "  step 2: map that byte through a 256-entry table to a key index (";
+            key_selector_detail::append_uint(s, N);
+            s += " means no match):\n";
+            for (std::size_t b = 0; b < 256; ++b) {
+                if (window.window_to_key[b] < N) {
+                    s += "    byte ";
+                    key_selector_detail::append_byte(s, static_cast<unsigned>(b));
+                    s += " -> key ";
+                    key_selector_detail::append_uint(s, window.window_to_key[b]);
+                    s += "\n";
+                }
+            }
+            s += "  step 3: confirm the candidate by checking the closing quote sits at the key's length and comparing the key bytes\n";
+        } else {
+            // Mirrors the perfect-hash path of match_raw().
+            if constexpr (phf.num_positions == key_selector_detail::HD_MODE) {
+                s += "algorithm: hash-and-displace perfect hash\n";
+                s += "  step 1: bucket = (first_byte + last_byte*3 + length*17) mod 256\n";
+                s += "  step 2: keyhash = base-31 rolling hash of the length and the first ";
+                key_selector_detail::append_uint(s, phf.hd_hash_variant);
+                s += " bytes\n";
+                s += "  step 3: slot = (displacement[bucket] + keyhash) mod ";
+                key_selector_detail::append_uint(s, table_size);
+                s += "\n  step 4: slot_to_key[slot] gives the candidate key index\n";
+                s += "  per-key derivation:\n";
+                for (std::size_t i = 0; i < N; ++i) {
+                    std::string_view k = keys[i];
+                    std::size_t bucket = key_selector_detail::hd_bucket_hash(k);
+                    std::size_t kh = (phf.hd_hash_variant == key_selector_detail::HD_HASH_2BYTE_FLAG)
+                        ? key_selector_detail::hd_key_hash_2(k)
+                        : key_selector_detail::hd_key_hash_4(k);
+                    std::size_t slot = (phf.asso_values[0][bucket] + kh) & (table_size - 1);
+                    s += "    \"";
+                    s += k;
+                    s += "\": bucket=";
+                    key_selector_detail::append_uint(s, bucket);
+                    s += " displacement=";
+                    key_selector_detail::append_uint(s, phf.asso_values[0][bucket]);
+                    s += " keyhash=";
+                    key_selector_detail::append_uint(s, kh);
+                    s += " slot=";
+                    key_selector_detail::append_uint(s, slot);
+                    s += "\n";
+                }
+            } else {
+                s += "algorithm: gperf-style perfect hash over ";
+                key_selector_detail::append_uint(s, phf.num_positions);
+                s += " character position(s)\n";
+                s += "  step 1: h = key length\n";
+                s += "  step 2: for each position below, add its association value for the key byte there (a position past the key's length contributes 0):\n";
+                for (std::size_t i = 0; i < phf.num_positions; ++i) {
+                    s += "    position ";
+                    if (phf.positions[i] == key_selector_detail::POS_LAST_CHAR) {
+                        s += "last character";
+                    } else {
+                        s += "byte index ";
+                        key_selector_detail::append_uint(s, phf.positions[i]);
+                    }
+                    s += "\n";
+                }
+                s += "  step 3: slot = h mod ";
+                key_selector_detail::append_uint(s, table_size);
+                s += " (a power of two, applied as a bitmask)\n";
+                s += "  step 4: slot_to_key[slot] gives the candidate key index\n";
+                s += "  per-key derivation:\n";
+                for (std::size_t i = 0; i < N; ++i) {
+                    std::string_view k = keys[i];
+                    std::size_t h = k.size();
+                    for (std::size_t pi = 0; pi < phf.num_positions; ++pi) {
+                        std::size_t pos = phf.positions[pi];
+                        std::size_t idx = (pos == key_selector_detail::POS_LAST_CHAR)
+                                          ? (k.size() - 1) : pos;
+                        if (idx < k.size()) {
+                            h += phf.asso_values[pi][static_cast<unsigned char>(k[idx])];
+                        }
+                    }
+                    std::size_t slot = h & (table_size - 1);
+                    s += "    \"";
+                    s += k;
+                    s += "\": h=";
+                    key_selector_detail::append_uint(s, h);
+                    s += " slot=";
+                    key_selector_detail::append_uint(s, slot);
+                    s += "\n";
+                }
+            }
+            s += "  occupied slots (slot -> key):\n";
+            for (std::size_t slot = 0; slot < table_size; ++slot) {
+                if (phf.slot_to_key[slot] < N) {
+                    s += "    slot ";
+                    key_selector_detail::append_uint(s, slot);
+                    s += " -> key ";
+                    key_selector_detail::append_uint(s, phf.slot_to_key[slot]);
+                    s += " (\"";
+                    s += keys[phf.slot_to_key[slot]];
+                    s += "\", length ";
+                    key_selector_detail::append_uint(s, phf.slot_key_len[slot]);
+                    s += ")\n";
+                }
+            }
+            s += "  confirm the candidate by checking the key length matches and comparing the key bytes\n";
+        }
+        return s;
+    }
+};
+
+namespace key_selector_detail {
+template <typename> struct is_key_selector : std::false_type {};
+template <constevalutil::fixed_string... Keys>
+struct is_key_selector<key_selector<Keys...>> : std::true_type {};
+} // namespace key_selector_detail
+
+/**
+ * Matches any instantiation of key_selector<Keys...>. Used to constrain
+ * object::for_each so that passing a non-selector type yields a clear
+ * constraint error rather than a cascade of failures inside for_each.
+ */
+template <typename T>
+concept key_selector_type = key_selector_detail::is_key_selector<T>::value;
+
+} // namespace ondemand
+} // namespace lasx
+} // namespace simdjson
+
+#endif // SIMDJSON_SUPPORTS_CONCEPTS
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+/* end file simdjson/generic/ondemand/key_selector.h for lasx */
 /* including simdjson/generic/ondemand/object.h for lasx: #include "simdjson/generic/ondemand/object.h" */
 /* begin file simdjson/generic/ondemand/object.h for lasx */
 #ifndef SIMDJSON_GENERIC_ONDEMAND_OBJECT_H
@@ -164427,6 +210547,7 @@ public:
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/implementation_simdjson_result_base.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/key_selector.h" */
 /* amalgamation skipped (editor-only): #include <vector> */
 /* amalgamation skipped (editor-only): #if SIMDJSON_STATIC_REFLECTION && SIMDJSON_SUPPORTS_CONCEPTS */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_string_builder.h"  // for constevalutil::fixed_string */
@@ -164437,6 +210558,114 @@ namespace simdjson {
 namespace lasx {
 namespace ondemand {

+#if SIMDJSON_SUPPORTS_CONCEPTS
+/**
+ * Result of object::for_each: the first error encountered (SUCCESS if none) and
+ * the number of distinct selector keys that matched during the walk. A
+ * matched_count equal to Selector::size() means every selected key was present
+ * in the object. Implicitly converts to error_code so existing callers that only
+ * care about the error (including SIMDJSON_TRY and the test ASSERT_* macros) keep
+ * working unchanged.
+ *
+ * On error paths (a handler returns a non-SUCCESS error_code, or a direct-target
+ * deserialization via value::get fails), matched_count reflects only the number
+ * of distinct selected keys that were successfully handled *before* the failure;
+ * the failing key is not counted. This is the current behavior.
+ *
+ * Marked [[nodiscard]]: for_each surfaces parse/type errors (from a matched
+ * value, a returning handler, or the structural walk itself) only through this
+ * result, so it must not be silently dropped. This matters most in builds
+ * without exceptions, where it is the only channel for those errors. To
+ * deliberately ignore it, assign to a variable or cast to void.
+ */
+struct [[nodiscard]] for_each_result {
+  error_code error{SUCCESS};
+  std::size_t matched_count{0};
+  constexpr operator error_code() const noexcept { return error; }
+};
+
+namespace key_selector_for_each_detail {
+/**
+ * A per-key handler for the variadic object::for_each is either:
+ *   - an invocable taking a value (run custom logic for that field), or
+ *   - a deserialization target T, in which case the matched value is assigned
+ *     directly via value::get(T&).
+ * The target form lets callers bind fields straight to variables without a
+ * lambda per key -- e.g. obj.for_each<"name","city","age">(name, city, age) --
+ * while still allowing a lambda in any position when custom logic is needed
+ * (for example to descend into a nested object).
+ */
+template <typename H>
+concept field_handler =
+    std::is_invocable_v<std::remove_reference_t<H>&, value> ||
+    ::simdjson::deserializable<std::remove_cvref_t<H>, value>;
+
+/**
+ * noexcept-ness of handling a single handler: nothrow-invocability for the
+ * callback form, or nothrow-deserializability for the direct-target form
+ * (builtin scalar/string targets are noexcept; custom targets follow their
+ * own tag_invoke noexcept specification).
+ */
+template <typename H>
+inline constexpr bool nothrow_field_handler_v =
+    std::is_invocable_v<std::remove_reference_t<H>&, value>
+        ? std::is_nothrow_invocable_v<std::remove_reference_t<H>&, value>
+        : ::simdjson::nothrow_deserializable<std::remove_cvref_t<H>, value>;
+
+/**
+ * nothrow_field_handler_v folded over a whole handler pack. The variadic
+ * for_each overloads spell their conditional-noexcept specifier with this
+ * single id-expression rather than an inline fold expression. MSVC (VS18)
+ * mishandles a fold-expression that appears directly in the noexcept-specifier
+ * of an out-of-line template definition: it fails to recognize the definition's
+ * specifier as matching the in-class declaration's and reports a spurious
+ * C2382 ("redefinition; different exception specifications"). Naming the fold
+ * here keeps the specifier a plain identifier, which matches reliably.
+ */
+template <typename... Handlers>
+inline constexpr bool nothrow_field_handlers_v =
+    (nothrow_field_handler_v<Handlers> && ...);
+} // namespace key_selector_for_each_detail
+#endif
+
+/**
+ * An opaque snapshot of an object's scanning position, obtained from
+ * object::get_current_position() and consumed by object::revert_position().
+ *
+ * This bundles the raw token position with the iteration depth that was
+ * live at the moment of capture: a scalar field value (e.g. a number)
+ * unwinds back to the object's own depth once fully consumed, while a
+ * compound value (array/object) not yet fully consumed stays one level
+ * deeper. The correct depth to restore to therefore depends on what was
+ * live when the snapshot was taken, not on the object itself, which is
+ * why both pieces travel together here rather than being recomputed later.
+ *
+ * The members are private and only constructible by object itself: a
+ * hand-built value here would let revert_position() jump to an arbitrary,
+ * unvalidated position and depth, and this type carries no way to check
+ * that a given snapshot still refers to a live object. Only ever pass
+ * along a value you got from get_current_position(), and only back to the
+ * same object, before that object has been reset() or the parser has
+ * iterate()d a new document: neither invalidates a snapshot in a way this
+ * type can detect.
+ */
+struct object_position {
+  /**
+   * Default-constructed so a variable can be declared and assigned later,
+   * matching e.g. document()/object(). Not a valid position to revert to.
+   */
+  simdjson_inline object_position() noexcept = default;
+
+private:
+  token_position position{};
+  depth_t depth{};
+
+  simdjson_inline object_position(token_position position_, depth_t depth_) noexcept
+    : position(position_), depth(depth_) {}
+
+  friend class object;
+};
+
 /**
  * A forward-only JSON object field iterator.
  */
@@ -164455,8 +210684,19 @@ public:
    * Using the iterator directly is also possible but error-prone and discouraged. In particular,
    * you must dereference the iterator exactly once per iteration (before calling '++').
    * Doing otherwise is unsafe and may lead to errors. You are responsible for ensuring
+   *
+   * With SIMDJSON_DEVELOPMENT_CHECKS, the iterator locks this object while it is
+   * alive, so that reentrant access (find_field(), reset(), ...) is reported as
+   * OUT_OF_ORDER_ITERATION.
+   */
+  simdjson_inline simdjson_result<object_iterator> begin() & noexcept;
+  /**
+   * Get an iterator to the start of a temporary object, e.g., `v.get_object().begin()`.
+   *
+   * The iterator does not depend on the object instance and may outlive it, so
+   * it does not lock it.
    */
-  simdjson_inline simdjson_result<object_iterator> begin() noexcept;
+  simdjson_inline simdjson_result<object_iterator> begin() && noexcept;
   simdjson_inline simdjson_result<object_iterator> end() noexcept;
   /**
    * Look up a field by name on an object (order-sensitive). By order-sensitive, we mean that
@@ -164468,10 +210708,11 @@ public:
    *
    * ```cpp
    * simdjson::ondemand::parser parser;
-   * auto obj = parser.parse(R"( { "x": 1, "y": 2, "z": 3 } )"_padded);
-   * double z = obj.find_field("z");
-   * double y = obj.find_field("y");
-   * double x = obj.find_field("x");
+   * auto json = R"( { "x": 1, "y": 2, "z": 3 } )"_padded;
+   * auto doc = parser.iterate(json);
+   * double z = doc.find_field("z");
+   * double y = doc.find_field("y");
+   * double x = doc.find_field("x");
    * ```
    * If you have multiple fields with a matching key ({"x": 1,  "x": 1}) be mindful
    * that only one field is returned.
@@ -164544,6 +210785,100 @@ public:
   /** @overload simdjson_inline simdjson_result<value> find_field_unordered(std::string_view key) & noexcept; */
   simdjson_inline simdjson_result<value> operator[](std::string_view key) && noexcept;

+#if SIMDJSON_SUPPORTS_CONCEPTS
+  /**
+   * Walk this object once and invoke on_match(selector_index, value) for each
+   * field whose key is in the compile-time key_selector Selector, in JSON order
+   * (first occurrence of a duplicate key wins). Iteration stops once all
+   * Selector::size() keys have matched or the object ends. The value is consumed
+   * in place, so this is a low-overhead way to extract a known set of fields
+   * regardless of their order in the JSON.
+   *
+   * Like other object iteration in simdjson, for_each consumes the object by
+   * advancing the underlying iterator state; after the call the same object
+   * instance should not be used for further field access or iteration.
+   *
+   * Usage:
+   *   using sel_t = ondemand::key_selector<"id", "text", "user">;
+   *   obj.for_each<sel_t>([&](std::size_t i, ondemand::value v) {
+   *     switch (i) { case 0: ...; case 1: ...; }
+   *   });
+   *
+   * Limitations (see key_selector): each key must be at most 63 characters long,
+   * and the number of keys should be moderate (hard limit 255; a handful is
+   * best, as the compile-time perfect hash may fail or slow compilation for
+   * large key sets). The keys must be distinct, non-empty, and free of backslash, double-quote and
+   * null bytes.
+   *
+   * The callback may return either void or an error_code. When it returns an
+   * error_code, the walk stops at the first non-SUCCESS result and that error is
+   * returned, which lets the callback surface value-parse errors.
+   *
+   * This function is conditionally noexcept: it is noexcept exactly when invoking
+   * the callback is noexcept. The callback runs inside this frame, so a throwing
+   * callback (e.g. one using the exception-throwing conversions like
+   * std::string_view(value) or uint64_t(value)) makes for_each potentially
+   * throwing too -- the exception propagates to the caller instead of crossing a
+   * noexcept boundary and calling std::terminate.
+   *
+   * @returns a for_each_result holding the first error encountered while walking
+   *          the object (including any error returned by the callback, SUCCESS if
+   *          none) and the number of distinct selector keys that matched. The
+   *          result converts implicitly to error_code, so callers that only need
+   *          the error can ignore the count.
+   */
+  template <typename Selector, typename Func>
+    requires key_selector_type<Selector> &&
+             std::is_invocable_v<Func&, std::size_t, value>
+  simdjson_inline for_each_result for_each(Func&& on_match)
+      noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, value>);
+
+  /**
+   * Variadic per-key form. Provide exactly one handler per key in the Selector
+   * (compiler-enforced). Handlers are processed in JSON document order for the
+   * matching keys. Each handler is either:
+   *   - a deserialization target (a variable), in which case the matched value
+   *     is assigned to it via value::get -- no lambda required; or
+   *   - an invocable taking the ondemand::value (for custom logic such as
+   *     descending into a nested object). It may return void or error_code;
+   *     returning error_code lets you surface parse/type errors.
+   * The two styles may be mixed freely, one handler per key.
+   *
+   * Example (bind fields straight to variables):
+   *   using fields = ondemand::key_selector<"name", "city", "age">;
+   *   obj.for_each<fields>(name, city, age);
+   *
+   * Example (mixing a target and a lambda):
+   *   obj.for_each<ondemand::key_selector<"id", "user">>(
+   *     id,                                          // assigned via value::get
+   *     [&](ondemand::value v){ u = read_user(v); }  // custom logic
+   *   );
+   *
+   * The index-based single-callback form (taking (size_t, value)) remains
+   * available for shared-state or more complex per-key logic.
+   */
+  template <typename Selector, typename... Handlers>
+    requires key_selector_type<Selector> &&
+             (sizeof...(Handlers) == Selector::size()) &&
+             (key_selector_for_each_detail::field_handler<Handlers> && ...)
+  simdjson_inline for_each_result for_each(Handlers&&... on_match)
+      noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
+  /**
+   * Direct-key shorthand. Equivalent to for_each<key_selector<Keys...>>(...).
+   * Lets you write the keys inline without a separate using/alias, binding each
+   * field straight to a variable (or a lambda, see the Selector form above):
+   *
+   *   obj.for_each<"name", "city", "age">(name, city, age);
+   */
+  template <constevalutil::fixed_string... Keys, typename... Handlers>
+    requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+             (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+             (key_selector_for_each_detail::field_handler<Handlers> && ...)
+  simdjson_inline for_each_result for_each(Handlers&&... on_match)
+      noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+#endif
+
   /**
    * Get the value associated with the given JSON pointer. We use the RFC 6901
    * https://tools.ietf.org/html/rfc6901 standard, interpreting the current node
@@ -164620,6 +210955,34 @@ public:
    * @returns true if the object contains some elements (not empty)
    */
   inline simdjson_result<bool> reset() & noexcept;
+  /**
+   * Get an opaque token representing the object's current scanning position.
+   * Pass it to revert_position() to return to this exact point later, without
+   * paying the cost of a full reset() and re-scan from the beginning.
+   *
+   * A typical use is an optional field that may or may not be next: capture
+   * the position, attempt find_field(), and on NO_SUCH_FIELD, revert_position()
+   * instead of reset() so that fields already consumed are not rescanned.
+   *
+   * The returned token is only valid for this object, and only until it is
+   * reset() or the parser iterate()s a new document; using it after either
+   * is undefined behavior (see object_position).
+   *
+   * @returns An opaque position token.
+   */
+  simdjson_inline object_position get_current_position() const noexcept;
+  /**
+   * Return the object's scanning position to a snapshot previously obtained
+   * from get_current_position(). Unlike reset(), this does not rescan the
+   * object from the beginning: fields before the captured position remain
+   * consumed, and scanning resumes exactly where the snapshot was captured.
+   *
+   * @param position A snapshot previously returned by get_current_position(),
+   *        for this same object.
+   * @returns SUCCESS, or OUT_OF_ORDER_ITERATION if called during active
+   *          iteration (SIMDJSON_DEVELOPMENT_CHECKS builds only).
+   */
+  simdjson_inline error_code revert_position(object_position position) noexcept;
   /**
    * This method scans the beginning of the object and checks whether the
    * object is empty.
@@ -164665,7 +211028,7 @@ public:
    */
   template <typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out)
-     noexcept(custom_deserializable<T, object> ? nothrow_custom_deserializable<T, object> : true) {
+     noexcept(nothrow_gettable<T, object>) {
     static_assert(custom_deserializable<T, object>);
     return deserialize(*this, out);
   }
@@ -164677,7 +211040,7 @@ public:
    */
   template <typename T>
   simdjson_inline simdjson_result<T> get()
-    noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+    noexcept(nothrow_gettable<T, object>)
   {
     static_assert(std::is_default_constructible<T>::value, "The specified type is not default constructible.");
     T out{};
@@ -164729,10 +211092,18 @@ protected:
   simdjson_warn_unused simdjson_inline error_code find_field_raw(const std::string_view key) noexcept;

   value_iterator iter{};
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  bool locked{false};
+  simdjson_inline void set_locked(bool _locked) noexcept;
+#endif

   friend class value;
   friend class document;
   friend struct simdjson_result<object>;
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  friend class object_iterator;
+  friend struct simdjson_result<object_iterator>;
+#endif
 };

 } // namespace ondemand
@@ -164748,7 +211119,8 @@ public:
   simdjson_inline simdjson_result(error_code error) noexcept; ///< @private
   simdjson_inline simdjson_result() noexcept = default;

-  simdjson_inline simdjson_result<lasx::ondemand::object_iterator> begin() noexcept;
+  simdjson_inline simdjson_result<lasx::ondemand::object_iterator> begin() & noexcept;
+  simdjson_inline simdjson_result<lasx::ondemand::object_iterator> begin() && noexcept;
   simdjson_inline simdjson_result<lasx::ondemand::object_iterator> end() noexcept;
   simdjson_inline simdjson_result<lasx::ondemand::value> find_field(std::string_view key) & noexcept;
   simdjson_inline simdjson_result<lasx::ondemand::value> find_field(std::string_view key) && noexcept;
@@ -164766,6 +211138,8 @@ public:
 #endif
   simdjson_inline error_code for_each_at_path_with_wildcard(std::string_view json_path, Func&& callback) noexcept;
   inline simdjson_result<bool> reset() noexcept;
+  inline simdjson_result<lasx::ondemand::object_position> get_current_position() noexcept;
+  inline error_code revert_position(lasx::ondemand::object_position position) noexcept;
   inline simdjson_result<bool> is_empty() noexcept;
   inline simdjson_result<size_t> count_fields() & noexcept;
   inline simdjson_result<std::string_view> raw_json() noexcept;
@@ -164773,7 +211147,7 @@ public:
   // TODO: move this code into object-inl.h

   template<typename T>
-  simdjson_inline simdjson_result<T> get() noexcept {
+  simdjson_inline simdjson_result<T> get() noexcept(nothrow_gettable<T, lasx::ondemand::object>) {
     if (error()) { return error(); }
     if constexpr (std::is_same_v<T, lasx::ondemand::object>) {
       return first;
@@ -164781,7 +211155,7 @@ public:
     return first.get<T>();
   }
   template<typename T>
-  simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept {
+  simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept(nothrow_gettable<T, lasx::ondemand::object>) {
     if (error()) { return error(); }
     if constexpr (std::is_same_v<T, lasx::ondemand::object>) {
       out = first;
@@ -164791,6 +211165,39 @@ public:
     return SUCCESS;
   }

+  /**
+   * Forwards to object::for_each on the underlying object, so error-code-style
+   * chains (e.g. doc["x"].get_object()) can call for_each without first
+   * extracting the object. If this result holds an error, that error is returned
+   * (with a zero match count) and the callback is not invoked. See
+   * object::for_each for the semantics.
+   */
+  template <typename Selector, typename Func>
+    requires lasx::ondemand::key_selector_type<Selector> &&
+             std::is_invocable_v<Func&, std::size_t, lasx::ondemand::value>
+  simdjson_inline lasx::ondemand::for_each_result for_each(Func&& on_match)
+      noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, lasx::ondemand::value>);
+
+  /**
+   * Forwarding overload for the variadic per-key form (targets and/or lambdas).
+   */
+  template <typename Selector, typename... Handlers>
+    requires lasx::ondemand::key_selector_type<Selector> &&
+             (sizeof...(Handlers) == Selector::size()) &&
+             (lasx::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+  simdjson_inline lasx::ondemand::for_each_result for_each(Handlers&&... on_match)
+      noexcept(lasx::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
+  /**
+   * Forwarding overload for the direct-key variadic form.
+   */
+  template <constevalutil::fixed_string... Keys, typename... Handlers>
+    requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+             (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+             (lasx::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+  simdjson_inline lasx::ondemand::for_each_result for_each(Handlers&&... on_match)
+      noexcept(lasx::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
 #if SIMDJSON_STATIC_REFLECTION
   // TODO: move this code into object-inl.h
   template<constevalutil::fixed_string... FieldNames, typename T>
@@ -164831,6 +211238,15 @@ public:
    */
   simdjson_inline object_iterator() noexcept = default;

+#if SIMDJSON_DEVELOPMENT_CHECKS
+   simdjson_inline ~object_iterator() noexcept;
+
+   simdjson_inline object_iterator(object_iterator&&) noexcept;
+   simdjson_inline object_iterator& operator=(object_iterator&&) noexcept;
+   simdjson_inline object_iterator(const object_iterator&) noexcept;
+   simdjson_inline object_iterator& operator=(const object_iterator&) noexcept;
+#endif
+
   //
   // Iterator interface
   //
@@ -164850,6 +211266,9 @@ public:
 private:
 #if SIMDJSON_DEVELOPMENT_CHECKS
    bool has_been_referenced{false};
+   object* parent{nullptr};
+
+   simdjson_inline object_iterator(const value_iterator &_iter, object* _parent) noexcept;
 #endif
   /**
    * The underlying JSON iterator.
@@ -164895,6 +211314,191 @@ public:

 #endif // SIMDJSON_GENERIC_ONDEMAND_OBJECT_ITERATOR_H
 /* end file simdjson/generic/ondemand/object_iterator.h for lasx */
+/* including simdjson/generic/ondemand/ranges.h for lasx: #include "simdjson/generic/ondemand/ranges.h" */
+/* begin file simdjson/generic/ondemand/ranges.h for lasx */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_RANGES_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_ONDEMAND_RANGES_H */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/field.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#if SIMDJSON_SUPPORTS_RANGES
+
+namespace simdjson {
+namespace lasx {
+namespace ondemand {
+
+/**
+ * A ranges-compatible iterator adapter for JSON arrays.
+ *
+ * Wraps array_iterator to satisfy std::input_iterator by providing:
+ * - const operator* (via mutable internal state)
+ * - post-increment operator
+ * - iterator_concept tag
+ *
+ * The mutable approach is standard for single-pass input iterators that
+ * read from external sources (similar to std::istream_iterator).
+ */
+class array_range_iterator {
+public:
+  using iterator_concept = std::input_iterator_tag;
+  using value_type = simdjson_result<value>;
+  using reference = simdjson_result<value>;
+  using difference_type = std::ptrdiff_t;
+
+  simdjson_inline array_range_iterator() noexcept = default;
+  simdjson_inline explicit array_range_iterator(array_iterator iter) noexcept;
+
+  /**
+   * Get the current element. Const-qualified for std::indirectly_readable;
+   * internally delegates to the mutable wrapped iterator.
+   */
+  simdjson_inline simdjson_result<value> operator*() const noexcept;
+  simdjson_inline array_range_iterator& operator++() noexcept;
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+  simdjson_inline void operator++(int) noexcept;
+SIMDJSON_POP_DISABLE_WARNINGS
+  /**
+   * Comparison delegates to array_iterator::operator==, which checks
+   * whether the underlying parser has finished the array (depth-based).
+   */
+  simdjson_inline friend bool operator==(const array_range_iterator& a,
+                                         const array_range_iterator& b) noexcept {
+    return a.iter_ == b.iter_;
+  }
+
+private:
+  mutable array_iterator iter_{};
+};
+
+/**
+ * A std::ranges::view over a JSON array.
+ *
+ * Wraps an ondemand::array and exposes begin()/end() that return
+ * array_range_iterator (satisfying std::input_iterator), enabling
+ * use with std::views::transform and other range adaptors.
+ *
+ * If the array's begin() returns an error (only possible under
+ * SIMDJSON_DEVELOPMENT_CHECKS), the range will be empty and error()
+ * will return the error code.
+ *
+ * Usage:
+ *   ondemand::parser parser;
+ *   auto doc = parser.iterate(json);
+ *   auto arr = doc.get_array().value();
+ *   for (auto elem : ondemand::get_range(arr)) { ... }
+ */
+class array_range {
+public:
+  simdjson_inline array_range() noexcept = default;
+  simdjson_inline explicit array_range(array& arr) noexcept;
+
+  simdjson_inline array_range_iterator begin() noexcept;
+  simdjson_inline array_range_iterator end() noexcept;
+
+  /** Returns SUCCESS if the range was created successfully, or the error code otherwise. */
+  simdjson_inline error_code error() const noexcept { return error_; }
+
+private:
+  array_iterator begin_{};
+  array_iterator end_{};
+  error_code error_{SUCCESS};
+};
+
+/**
+ * A ranges-compatible iterator adapter for JSON objects.
+ *
+ * Wraps object_iterator to satisfy std::input_iterator, yielding
+ * simdjson_result<field> elements (key-value pairs).
+ */
+class object_range_iterator {
+public:
+  using iterator_concept = std::input_iterator_tag;
+  using value_type = simdjson_result<field>;
+  using reference = simdjson_result<field>;
+  using difference_type = std::ptrdiff_t;
+
+  simdjson_inline object_range_iterator() noexcept = default;
+  simdjson_inline explicit object_range_iterator(object_iterator iter) noexcept;
+
+  simdjson_inline simdjson_result<field> operator*() const noexcept;
+  simdjson_inline object_range_iterator& operator++() noexcept;
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+  simdjson_inline void operator++(int) noexcept;
+SIMDJSON_POP_DISABLE_WARNINGS
+  simdjson_inline friend bool operator==(const object_range_iterator& a,
+                                         const object_range_iterator& b) noexcept {
+    return a.iter_ == b.iter_;
+  }
+
+private:
+  mutable object_iterator iter_{};
+};
+
+/**
+ * A std::ranges::view over a JSON object.
+ *
+ * Wraps an ondemand::object and exposes begin()/end() that return
+ * object_range_iterator, enabling use with range adaptors.
+ *
+ * If the object's begin() returns an error, the range will be empty
+ * and error() will return the error code.
+ */
+class object_range {
+public:
+  simdjson_inline object_range() noexcept = default;
+  simdjson_inline explicit object_range(object& obj) noexcept;
+
+  simdjson_inline object_range_iterator begin() noexcept;
+  simdjson_inline object_range_iterator end() noexcept;
+
+  /** Returns SUCCESS if the range was created successfully, or the error code otherwise. */
+  simdjson_inline error_code error() const noexcept { return error_; }
+
+private:
+  object_iterator begin_{};
+  object_iterator end_{};
+  error_code error_{SUCCESS};
+};
+
+/** Get a std::ranges compatible view over a JSON array. */
+simdjson_inline array_range get_range(array& arr) noexcept;
+
+/** Get a std::ranges compatible view over a JSON object (key-value pairs). */
+simdjson_inline object_range get_key_value_range(object& obj) noexcept;
+
+#if SIMDJSON_EXCEPTIONS
+/** Get a std::ranges compatible view, unwrapping the simdjson_result (throws on error). */
+simdjson_inline array_range get_range(simdjson_result<array> result);
+
+/** Get a std::ranges compatible view, unwrapping the simdjson_result (throws on error). */
+simdjson_inline object_range get_key_value_range(simdjson_result<object> result);
+#endif // SIMDJSON_EXCEPTIONS
+
+} // namespace ondemand
+} // namespace lasx
+} // namespace simdjson
+
+namespace std {
+namespace ranges {
+template<>
+inline constexpr bool enable_view<simdjson::lasx::ondemand::array_range> = true;
+template<>
+inline constexpr bool enable_view<simdjson::lasx::ondemand::object_range> = true;
+} // namespace ranges
+} // namespace std
+
+#endif // SIMDJSON_SUPPORTS_RANGES
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_RANGES_H
+/* end file simdjson/generic/ondemand/ranges.h for lasx */
 /* including simdjson/generic/ondemand/serialization.h for lasx: #include "simdjson/generic/ondemand/serialization.h" */
 /* begin file simdjson/generic/ondemand/serialization.h for lasx */
 #ifndef SIMDJSON_GENERIC_ONDEMAND_SERIALIZATION_H
@@ -165027,12 +211631,14 @@ inline std::ostream& operator<<(std::ostream& out, simdjson::simdjson_result<sim
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/annotations.h" */
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */

 #include <concepts>
 #include <limits>
 #if SIMDJSON_STATIC_REFLECTION
 #include <meta>
+#include <vector>
 // #include <static_reflection> // for std::define_static_string - header not available yet
 #endif

@@ -165057,10 +211663,25 @@ error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept {

 template <std::floating_point T>
 error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept {
-  double x;
-  SIMDJSON_TRY(val.get_double().get(x));
-  out = static_cast<T>(x);
-  return SUCCESS;
+  if constexpr (std::is_same_v<T, float>) {
+    // Going through binary64 and then rounding to binary32 would round twice
+    // and could produce a value that is not the float nearest to the JSON
+    // number, so we parse to binary32 directly.
+    return val.get_float().get(out);
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  } else if constexpr (std::is_same_v<T, std::float32_t>) {
+    // Same reason as float.
+    float x;
+    SIMDJSON_TRY(val.get_float().get(x));
+    out = static_cast<T>(x);
+    return SUCCESS;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+  } else {
+    double x;
+    SIMDJSON_TRY(val.get_double().get(x));
+    out = static_cast<T>(x);
+    return SUCCESS;
+  }
 }

 template <std::signed_integral T>
@@ -165096,11 +211717,59 @@ template <concepts::constructible_from_string_view T, typename ValT>
 error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(std::is_nothrow_constructible_v<T, std::string_view>) {
   std::string_view str;
   SIMDJSON_TRY(val.get_string().get(str));
-  out = T{str};
+  if constexpr (requires { out.assign(str.data(), str.size()); }) {
+    // Copy straight into out (e.g., std::string): building a temporary and
+    // move-assigning it is markedly slower.
+    out.assign(str.data(), str.size());
+  } else {
+    out = T{str};
+  }
+  return SUCCESS;
+}
+
+#if SIMDJSON_SUPPORTS_CHAR8_T
+// any C++20 char8_t string-like type (can be constructed from std::u8string_view),
+// such as std::u8string
+template <concepts::constructible_from_u8string_view T, typename ValT>
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(std::is_nothrow_constructible_v<T, std::u8string_view>) {
+  std::u8string_view str;
+  SIMDJSON_TRY(val.get_u8string().get(str));
+  if constexpr (requires { out.assign(str.data(), str.size()); }) {
+    // Copy straight into out (e.g., std::u8string), as for std::string above.
+    out.assign(str.data(), str.size());
+  } else {
+    out = T{str};
+  }
   return SUCCESS;
 }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T


+namespace details {
+// Whether to deserialize the elements of the container T directly into a new
+// element (emplace_one(out) and then get) rather than into a temporary that is
+// then moved into the container. A temporary of a trivially copyable type lives
+// in registers and is cheap to move, and value-initializing a small element in
+// place costs more than it saves (GCC zeroes it with rep stos). But moving a
+// large element, or a string (whose deserialization then needs a temporary
+// string and a move assignment of its own), is expensive.
+template <typename T>
+concept deserialize_in_place =
+    concepts::returns_reference<T> && requires(T &c) { c.pop_back(); } &&
+    !std::is_trivially_copyable_v<typename T::value_type> &&
+    (sizeof(typename T::value_type) > 32 || concepts::constructible_from_string_view<typename T::value_type>);
+
+// Removes the last element of the container on destruction while armed.
+template <typename T>
+struct pop_back_guard {
+  T &container;
+  bool armed{true};
+  ~pop_back_guard() {
+    if (armed) { container.pop_back(); }
+  }
+};
+} // namespace details
+
 /**
  * STL containers have several constructors including one that takes a single
  * size argument. Thus, some compilers (Visual Studio) will not be able to
@@ -165124,22 +211793,65 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
     SIMDJSON_TRY(val.get_array().get(arr));
   }

-  for (auto v : arr) {
-    if constexpr (concepts::returns_reference<T>) {
-      if (auto const err = v.get<value_type>().get(concepts::emplace_one(out));
-          err) {
-        // If an error occurs, the empty element that we just inserted gets
-        // removed. We're not using a temp variable because if T is a heavy
-        // type, we want the valid path to be the fast path and the slow path be
-        // the path that has errors in it.
-        if constexpr (requires { out.pop_back(); }) {
-          static_cast<void>(out.pop_back());
+  if constexpr (std::is_same_v<T, std::vector<value_type>> && !std::is_same_v<value_type, bool>) {
+    // Collect the elements in a per-thread scratch vector that keeps its
+    // capacity from call to call, then move them into out after reserving the
+    // exact size: out is allocated once instead of being regrown. A nested
+    // array of the same type finds the scratch busy and takes the paths below.
+    // Prior related work: jsonifier keeps a thread-local vector and sizes the
+    // caller's vector from that element count (parse_impl.hpp,
+    // https://github.com/nihilai-collective/Jsonifier).
+    struct scratch_space {
+      std::vector<value_type> elements{};
+      bool busy{false};
+    };
+    static thread_local scratch_space scratch;
+    if (!scratch.busy && out.empty()) {
+      struct release_scratch {
+        scratch_space &s;
+        T &out;
+        size_t parsed{0};
+        bool complete{false};
+        // On an error or an exception, out gets the elements parsed so far (as
+        // with the loops below), without allocating. Kept out of the hot path.
+        simdjson_never_inline void keep_parsed() noexcept {
+          s.elements.resize(parsed);
+          out.swap(s.elements);
         }
-        return err;
-      }
-    } else {
+        ~release_scratch() {
+          if (simdjson_unlikely(!complete)) { keep_parsed(); }
+          s.elements.clear();
+          // Do not hold on to the memory of a very large array.
+          if (s.elements.capacity() * sizeof(value_type) > (1 << 20)) { std::vector<value_type>().swap(s.elements); }
+          s.busy = false;
+        }
+      } release{scratch, out};
+      scratch.busy = true;
+      for (auto v : arr) {
+        SIMDJSON_TRY(v.get<value_type>(scratch.elements.emplace_back()));
+        release.parsed++;
+      }
+      out.reserve(release.parsed);
+      release.complete = true;
+      for (auto &e : scratch.elements) { out.emplace_back(std::move(e)); }
+      return SUCCESS;
+    }
+  }
+  if constexpr (details::deserialize_in_place<T>) {
+    for (auto v : arr) {
+      auto &slot = concepts::emplace_one(out);
+      // An error or an exception (a user tag_invoke may throw) must not leave
+      // a partially deserialized element behind.
+      details::pop_back_guard<T> guard{out};
+      SIMDJSON_TRY(v.get<value_type>(slot));
+      guard.armed = false;
+    }
+  } else {
+    for (auto v : arr) {
+      // Deserialize into a temporary first: an error or an exception (a user
+      // tag_invoke may throw) must not leave a default-constructed element behind.
       value_type temp;
-      if (auto const err = v.get<value_type>().get(temp); err) {
+      if (auto const err = v.get<value_type>(temp); err) {
         return err;
       }
       concepts::emplace_one(out, std::move(temp));
@@ -165180,7 +211892,7 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
 }

 template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, lasx::ondemand::object &obj, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, lasx::ondemand::object &obj, T &out) noexcept(false) {
   using value_type = typename std::remove_cvref_t<T>::mapped_type;

   out.clear();
@@ -165199,21 +211911,21 @@ error_code tag_invoke(deserialize_tag, lasx::ondemand::object &obj, T &out) noex
 }

 template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, lasx::ondemand::value &val, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, lasx::ondemand::value &val, T &out) noexcept(false) {
   lasx::ondemand::object obj;
   SIMDJSON_TRY(val.get_object().get(obj));
   return simdjson::deserialize(obj, out);
 }

 template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, lasx::ondemand::document &doc, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, lasx::ondemand::document &doc, T &out) noexcept(false) {
   lasx::ondemand::object obj;
   SIMDJSON_TRY(doc.get_object().get(obj));
   return simdjson::deserialize(obj, out);
 }

 template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, lasx::ondemand::document_reference &doc, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, lasx::ondemand::document_reference &doc, T &out) noexcept(false) {
   lasx::ondemand::object obj;
   SIMDJSON_TRY(doc.get_object().get(obj));
   return simdjson::deserialize(obj, out);
@@ -165224,10 +211936,6 @@ error_code tag_invoke(deserialize_tag, lasx::ondemand::document_reference &doc,
  * This CPO (Customization Point Object) will help deserialize into
  * smart pointers.
  *
- * If constructing T is nothrow, this conversion should be nothrow as well since
- * we return MEMALLOC if we're not able to allocate memory instead of throwing
- * the error message.
- *
  * @tparam T The type inside the smart pointer
  * @tparam ValT document/value type
  * @param val document/value
@@ -165235,7 +211943,7 @@ error_code tag_invoke(deserialize_tag, lasx::ondemand::document_reference &doc,
  * @return status of the conversion
  */
 template <concepts::smart_pointer T, typename ValT>
-error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(nothrow_deserializable<typename std::remove_cvref_t<T>::element_type, ValT>) {
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
   using element_type = typename std::remove_cvref_t<T>::element_type;

   // For better error messages, don't use these as constraints on
@@ -165247,12 +211955,13 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(nothrow_deser
       std::is_default_constructible_v<element_type>,
       "The specified type inside the unique_ptr must default constructible.");

-  auto ptr = new (std::nothrow) element_type();
-  if (ptr == nullptr) {
+  // Own the allocation before get(): a user tag_invoke may throw.
+  std::unique_ptr<element_type> ptr(new (std::nothrow) element_type());
+  if (!ptr) {
     return MEMALLOC;
   }
   SIMDJSON_TRY(val.template get<element_type>(*ptr));
-  out.reset(ptr);
+  out = std::move(ptr);
   return SUCCESS;
 }

@@ -165284,53 +211993,595 @@ error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept(nothrow_deser

 template <typename T>
 constexpr bool user_defined_type = (std::is_class_v<T>
-&& !std::is_same_v<T, std::string> && !std::is_same_v<T, std::string_view> && !concepts::optional_type<T> &&
-!concepts::appendable_containers<T>);
+&& !std::is_same_v<T, std::string> && !std::is_same_v<T, std::string_view>
+#if SIMDJSON_SUPPORTS_CHAR8_T
+// The char8_t string types are class types with no reflectable members, so
+// without this they would be taken for user structs and the reflection
+// overload below would compete with the u8 string overload in
+// std_deserialize.h, making every tag_invoke call on them ambiguous.
+&& !std::is_same_v<T, std::u8string> && !std::is_same_v<T, std::u8string_view>
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+&& !concepts::optional_type<T> &&
+!concepts::appendable_containers<T>
+// simdjson's own types (array, object, value, raw_json_string, number,
+// document, document_reference) have dedicated get<T>() specializations and
+// must never go through reflection.
+&& !is_builtin_deserializable_v<T>
+&& !std::is_same_v<T, lasx::ondemand::number>
+&& !std::is_same_v<T, lasx::ondemand::document>
+&& !std::is_same_v<T, lasx::ondemand::document_reference>);
+
+
+// key_selector_reflection_detail is defined unconditionally (it only requires
+// static reflection). It provides the compile-time machinery for building a
+// key_selector from a struct's members, the per-member helpers shared by every
+// deserialization path (they implement the annotations of annotations.h), the
+// ordered per-member path used by the opt-out build (see
+// deserialize_struct_ordered below), and the scan used for structs annotated
+// with deny_unknown_fields and as an automatic fallback (see
+// deserialize_struct_scan and keys_fit_selector below).
+namespace key_selector_reflection_detail {
+
+// A member participates if it is public, non-const, and not annotated with skip
+// or skip_deserializing.
+consteval bool is_eligible_member(std::meta::info mem) {
+  return !std::meta::is_const(mem) && std::meta::is_public(mem)
+      && !simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_tag)
+      && !simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_deserializing_tag);
+}
+
+// A field is a data member reached from T through a chain of members: usually a
+// direct member of T (a path of length one), or a member of a member annotated
+// with flatten (the members of a flattened structure are fields of the enclosing
+// one). A field is identified by the type member_path<m1, m2, ..., leaf>.
+template <std::meta::info First, std::meta::info... Rest>
+struct member_path {
+  // The data member holding the value; its annotations drive (de)serialization.
+  static constexpr std::meta::info leaf = [] {
+    std::meta::info members[] = {First, Rest...};
+    return members[sizeof...(Rest)];
+  }();
+  template <typename T>
+  static simdjson_inline constexpr auto &get(T &obj) noexcept {
+    if constexpr (sizeof...(Rest) == 0) {
+      return obj.[:First:];
+    } else {
+      return member_path<Rest...>::get(obj.[:First:]);
+    }
+  }
+};
+
+// True when T can be flattened: a structure deserialized member by member (not
+// a string, a container, an optional, a smart pointer, ...).
+template <typename T>
+constexpr bool flattenable_type = user_defined_type<T> && !concepts::string_view_keyed_map<T>
+    && !concepts::container_but_not_string<T> && !concepts::smart_pointer<T>;
+
+consteval void append_eligible_fields(std::meta::info type, std::vector<std::meta::info> &prefix,
+                                      std::vector<std::meta::info> &fields) {
+  for (std::meta::info mem : std::meta::nonstatic_data_members_of(type, std::meta::access_context::unchecked())) {
+    if (!is_eligible_member(mem)) { continue; }
+    prefix.push_back(std::meta::reflect_constant(mem));
+    if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::flatten_tag)) {
+      std::meta::info flattened = simdjson::detail::flattened_type(mem);
+      if (!std::meta::extract<bool>(std::meta::substitute(^^flattenable_type, {flattened}))) {
+        throw std::meta::exception(u8"simdjson::flatten requires a member whose type is a structure deserialized member by member", mem);
+      }
+      append_eligible_fields(flattened, prefix, fields);
+    } else {
+      fields.push_back(std::meta::substitute(^^member_path, prefix));
+    }
+    prefix.pop_back();
+  }
+}
+
+// The fields of `type` that participate in deserialization, in declaration order
+// (the fields of a flattened member take its place). A field's position in this
+// list is its "field index" below.
+consteval std::vector<std::meta::info> eligible_fields(std::meta::info type) {
+  std::vector<std::meta::info> prefix;
+  std::vector<std::meta::info> fields;
+  append_eligible_fields(type, prefix, fields);
+  return fields;
+}
+
+// The data member at the end of a field path.
+consteval std::meta::info field_leaf(std::meta::info path) {
+  return std::meta::extract<std::meta::info>(std::meta::template_arguments_of(path).back());
+}
+
+// Number of fields that participate in deserialization. A class can have zero
+// eligible fields (e.g. std::chrono::time_point, whose only data member is
+// private): an empty key_selector cannot be built, so the tag_invoke below
+// special-cases this count.
+template <typename T>
+consteval std::size_t eligible_field_count() {
+  return eligible_fields(^^T).size();
+}
+
+// The keys accepted for a member (or an enumerator): its JSON key followed by its
+// aliases. They are static strings, so they can be used as template arguments.
+consteval std::vector<const char *> accepted_keys_of(std::meta::info entity) {
+  std::vector<const char *> keys;
+  for (std::string_view key : simdjson::detail::json_key_names(entity)) {
+    bool repeated = false;
+    for (const char *previous : keys) { repeated = repeated || std::string_view(previous) == key; }
+    if (!repeated) { keys.push_back(std::define_static_string(key)); }
+  }
+  return keys;
+}
+
+// Every key accepted for T: the eligible fields in order, each followed by its
+// aliases. This is the key list of T's key_selector.
+consteval std::vector<const char *> accepted_keys(std::meta::info type) {
+  std::vector<const char *> keys;
+  for (std::meta::info path : eligible_fields(type)) {
+    for (const char *key : accepted_keys_of(field_leaf(path))) { keys.push_back(key); }
+  }
+  return keys;
+}
+
+// The field index of each entry of accepted_keys(type).
+consteval std::vector<std::size_t> accepted_key_fields(std::meta::info type) {
+  std::vector<std::size_t> key_fields;
+  std::vector<std::meta::info> fields = eligible_fields(type);
+  for (std::size_t i = 0; i < fields.size(); ++i) {
+    for (std::size_t j = 0; j < accepted_keys_of(field_leaf(fields[i])).size(); ++j) { key_fields.push_back(i); }
+  }
+  return key_fields;
+}
+
+// True when no two eligible fields of T accept the same key. Otherwise one JSON
+// key would have to fill several members: this is reported at compile time.
+template <typename T>
+consteval bool accepted_keys_are_distinct() {
+  std::vector<const char *> keys = accepted_keys(^^T);
+  for (std::size_t i = 0; i < keys.size(); ++i) {
+    for (std::size_t j = i + 1; j < keys.size(); ++j) {
+      if (std::string_view(keys[i]) == std::string_view(keys[j])) { return false; }
+    }
+  }
+  return true;
+}
+
+// True when some accepted key of T is written with escape sequences in JSON (a
+// double quote, a backslash or a control character). Such a key can only be
+// matched by comparing unescaped keys (deserialize_struct_scan): obj[key] and
+// the key_selector compare the raw bytes.
+template <typename T>
+consteval bool keys_need_unescaping() {
+  for (std::string_view key : accepted_keys(^^T)) {
+    for (char c : key) {
+      if (c == '\\' || c == '"' || static_cast<unsigned char>(c) < 0x20) { return true; }
+    }
+  }
+  return false;
+}

+// True when some eligible field of T has aliases: several selector keys may then
+// map to the same field.
+template <typename T>
+consteval bool has_aliases() {
+  return accepted_keys(^^T).size() != eligible_fields(^^T).size();
+}
+
+// True when `mem` is annotated with default_value or default_from (directly, or
+// through a default_value annotation on the enclosing structure).
+template <auto mem>
+consteval bool has_default() {
+  return simdjson::detail::has_annotation(mem, ^^simdjson::detail::default_value_tag)
+      || simdjson::detail::has_annotation(std::meta::parent_of(mem), ^^simdjson::detail::default_value_tag)
+      || simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::default_from_t) != std::meta::info{};
+}
+
+// True when a missing key for `mem` is not an error: optional members and
+// members with a default.
+template <auto mem>
+consteval bool may_be_absent() {
+  return concepts::optional_type<typename [: std::meta::type_of(mem) :]> || has_default<mem>();
+}
+
+// True when every eligible field of T is required (none may be absent). In that
+// case presence can be checked with a single match count instead of a per-field
+// "seen" array.
+template <typename T>
+consteval bool all_eligible_fields_required() {
+  bool all_required = true;
+  template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+    if constexpr (may_be_absent<[: path :]::leaf>()) {
+      all_required = false;
+    }
+  }
+  return all_required;
+}
+
+// `key` as a constevalutil::fixed_string usable as an NTTP.
+template <const char *key>
+consteval auto key_fixed_string() {
+  constexpr std::string_view key_view{ key };
+  char buffer[key_view.size() + 1] = {};
+  for (std::size_t i = 0; i < key_view.size(); ++i) { buffer[i] = key_view[i]; }
+  return constevalutil::fixed_string<key_view.size() + 1>(buffer);
+}
+
+// key_selector template arguments (one fixed_string per accepted key), in the
+// order of accepted_keys.
+template <typename T>
+consteval std::vector<std::meta::info> selector_key_args() {
+  std::vector<std::meta::info> args;
+  template for (constexpr const char *key : std::define_static_array(accepted_keys(^^T))) {
+    args.push_back(std::meta::reflect_constant(key_fixed_string<key>()));
+  }
+  return args;
+}
+
+// key_selector whose keys are exactly T's accepted keys (index i <-> i-th key).
+template <typename T>
+using selector_for = typename [: std::meta::substitute(
+    ^^lasx::ondemand::key_selector, selector_key_args<T>()) :];
+
+// True when T's accepted keys satisfy every key_selector requirement, so a
+// key_selector can be built for T without a compile-time error. This mirrors the
+// key_selector limits (see key_selector.h): at most 255 keys, each key non-empty
+// and at most 63 characters, no backslash / double-quote / null byte, and all
+// keys distinct. Keys with any other control character are excluded too: they
+// are escaped in JSON, so their raw bytes never match. When this returns false
+// the deserializer falls back to deserialize_struct_scan instead of failing to
+// compile.
+template <typename T>
+consteval bool keys_fit_selector() {
+  std::vector<std::string_view> keys;
+  for (const char *key : accepted_keys(^^T)) { keys.push_back(key); }
+  if (keys.size() > 255) { return false; }
+  for (std::size_t i = 0; i < keys.size(); ++i) {
+    if (keys[i].empty() || keys[i].size() > 63) { return false; }
+    for (char c : keys[i]) {
+      if (c == '\\' || c == '"' || static_cast<unsigned char>(c) < 0x20) { return false; }
+    }
+    for (std::size_t j = i + 1; j < keys.size(); ++j) {
+      if (keys[i] == keys[j]) { return false; }
+    }
+  }
+  return true;
+}
+
+// True when the class `adapter` declares a member named deserialize.
+consteval bool declares_deserialize(std::meta::info adapter) {
+  for (std::meta::info m : std::meta::members_of(adapter, std::meta::access_context::unchecked())) {
+    if (std::meta::has_identifier(m) && std::meta::identifier_of(m) == "deserialize") { return true; }
+  }
+  return false;
+}
+
+// Deserialize a JSON value into `target`, the storage of member `mem`, through
+// the member's with<Adapter> annotation when the adapter provides a deserialize
+// function.
+template <auto mem, typename ValueT, typename M>
+simdjson_warn_unused simdjson_inline error_code deserialize_member_value(ValueT &field_value, M &target) noexcept(false) {
+  constexpr std::meta::info with_type = simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::with_t);
+  if constexpr (with_type != std::meta::info{}) {
+    using adapter = typename [: with_type :]::adapter;
+    using ondemand_value = lasx::ondemand::value;
+    if constexpr (requires { { adapter::deserialize(field_value, target) } -> std::convertible_to<error_code>; }) {
+      return adapter::deserialize(field_value, target);
+    } else if constexpr (requires(ondemand_value &v) { { adapter::deserialize(v, target) } -> std::convertible_to<error_code>; }
+                         && requires { field_value.get_value(); }) {
+      // A transparent structure read from a document: the adapter takes an
+      // ondemand::value. A scalar document cannot be viewed as a value, so it
+      // reports SCALAR_DOCUMENT_AS_VALUE (an adapter taking auto& receives the
+      // document itself and has no such limitation).
+      ondemand_value v;
+      SIMDJSON_TRY(field_value.get_value().get(v));
+      return adapter::deserialize(v, target);
+    } else {
+      static_assert(!declares_deserialize(^^adapter),
+                    "the deserialize function of a simdjson::with adapter must be callable as "
+                    "Adapter::deserialize(simdjson::ondemand::value &, T &) and return an error_code");
+      return field_value.get(target);
+    }
+  } else {
+    return field_value.get(target);
+  }
+}
+
+// Deserialize the storage `target` of member `mem` from a JSON value.
+template <auto mem, typename ValueT, typename M>
+simdjson_warn_unused simdjson_inline error_code deserialize_member(ValueT &field_value, M &target) noexcept(false) {
+  if constexpr (has_default<mem>() && std::is_default_constructible_v<M> && std::is_move_assignable_v<M>) {
+    // A present key replaces the default value: deserialize into a fresh
+    // temporary so that, e.g., a container does not append to its default
+    // content, and a failure leaves the default untouched.
+    M value{};
+    SIMDJSON_TRY(deserialize_member_value<mem>(field_value, value));
+    target = std::move(value);
+    return SUCCESS;
+  } else {
+    return deserialize_member_value<mem>(field_value, target);
+  }
+}

+// Deserialize the field with the given field index.
+template <typename T>
+simdjson_warn_unused simdjson_inline error_code deserialize_field_at(
+    std::size_t field_index, lasx::ondemand::value field_value, T &out) noexcept(false) {
+  std::size_t counter = 0;
+  template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+    using field = [: path :];
+    if (field_index == counter) { return deserialize_member<field::leaf>(field_value, field::get(out)); }
+    ++counter;
+  }
+  return SUCCESS;
+}
+
+// Called when the key(s) of member `mem`, stored in `target`, are missing from
+// the JSON object: an error for a required member, a call to the default_from
+// factory, or nothing at all.
+template <auto mem, typename M>
+simdjson_warn_unused simdjson_inline error_code on_missing_member(M &target) noexcept(false) {
+  constexpr std::meta::info default_from_type = simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::default_from_t);
+  if constexpr (default_from_type != std::meta::info{}) {
+    target = [: default_from_type :]::factory();
+    return SUCCESS;
+  } else if constexpr (may_be_absent<mem>()) {
+    // For optional and default_value members, a missing key is not an error:
+    // leave the member at its current (default) value.
+    (void)target;
+    return SUCCESS;
+  } else {
+    (void)target;
+    return NO_SUCH_FIELD;
+  }
+}
+
+// Report or handle every field that was not seen in the JSON object.
+template <typename T, std::size_t N>
+simdjson_warn_unused simdjson_inline error_code handle_missing_fields(
+    const std::array<bool, N> &seen_field, T &out) noexcept(false) {
+  std::size_t counter = 0;
+  template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+    using field = [: path :];
+    if (!seen_field[counter]) { SIMDJSON_TRY(on_missing_member<field::leaf>(field::get(out))); }
+    ++counter;
+  }
+  return SUCCESS;
+}
+
+// Ordered, per-field deserialization: one obj[key] lookup per eligible field
+// (and per alias, until one is found). This is the opt-out path
+// (-DSIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION=1), except for structs with keys
+// that need unescaping (see keys_need_unescaping).
+template <typename T>
+simdjson_warn_unused error_code deserialize_struct_ordered(
+    lasx::ondemand::object &obj, T &out) noexcept(false) {
+  template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+    using field = [: path :];
+    lasx::ondemand::value field_value;
+    error_code error = NO_SUCH_FIELD;
+    template for (constexpr const char *key : std::define_static_array(accepted_keys_of(field::leaf))) {
+      if (error == NO_SUCH_FIELD) { error = obj[std::string_view(key)].get(field_value); }
+    }
+    if (error == NO_SUCH_FIELD) {
+      SIMDJSON_TRY(on_missing_member<field::leaf>(field::get(out)));
+    } else if (error) {
+      return error;
+    } else {
+      SIMDJSON_TRY(deserialize_member<field::leaf>(field_value, field::get(out)));
+    }
+  }
+  return SUCCESS;
+}
+
+// Appends the JSON keys that serialization writes for members of `type` that
+// deserialization cannot assign (const or non-public members, directly or
+// through flatten). `all` is true inside a flattened member that is itself
+// unassignable. Keys of skip_deserializing members are not included: they are
+// unknown keys, as documented.
+consteval void append_unassignable_keys(std::meta::info type, bool all, std::vector<const char *> &keys) {
+  for (std::meta::info mem : std::meta::nonstatic_data_members_of(type, std::meta::access_context::unchecked())) {
+    if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_tag)
+        || simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_serializing_tag)
+        || simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_deserializing_tag)) {
+      continue;
+    }
+    bool unassignable = all || !is_eligible_member(mem);
+    if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::flatten_tag)) {
+      append_unassignable_keys(simdjson::detail::flattened_type(mem), unassignable, keys);
+    } else if (unassignable) {
+      keys.push_back(std::define_static_string(simdjson::detail::json_key_name(mem)));
+    }
+  }
+}
+
+consteval std::vector<const char *> unassignable_keys(std::meta::info type) {
+  std::vector<const char *> keys;
+  append_unassignable_keys(type, false, keys);
+  return keys;
+}
+
+// Deserialization by a single pass over every field of the object, comparing
+// unescaped keys. It is used for structs annotated with deny_unknown_fields
+// (DenyUnknown = true), where a key that does not map to an eligible field is
+// reported as UNKNOWN_FIELD, and as the fallback for structs whose keys the
+// key_selector or obj[key] cannot match (see keys_fit_selector and
+// keys_need_unescaping). As with object::for_each, the first occurrence of a
+// field wins (later duplicates, or aliases of a field already seen, are ignored).
+template <bool DenyUnknown, typename T>
+simdjson_warn_unused error_code deserialize_struct_scan(
+    lasx::ondemand::object &obj, T &out) noexcept(false) {
+  static constexpr auto keys = std::define_static_array(accepted_keys(^^T));
+  static constexpr auto key_fields = std::define_static_array(accepted_key_fields(^^T));
+  std::array<bool, eligible_field_count<T>()> seen_field{};
+  for (auto field_result : obj) {
+    lasx::ondemand::field json_field;
+    SIMDJSON_TRY(std::move(field_result).get(json_field));
+    std::string_view key;
+    SIMDJSON_TRY(json_field.unescaped_key().get(key));
+    std::size_t key_index = keys.size();
+    for (std::size_t i = 0; i < keys.size(); ++i) {
+      if (key == std::string_view(keys[i])) { key_index = i; break; }
+    }
+    if (key_index == keys.size()) {
+      if constexpr (DenyUnknown) {
+        // A key that T itself serializes (e.g. of a const member) is not
+        // unknown: a serialized value must parse back.
+        static constexpr auto ignored_keys = std::define_static_array(unassignable_keys(^^T));
+        bool ignored = false;
+        for (const char *ignored_key : ignored_keys) {
+          if (key == std::string_view(ignored_key)) { ignored = true; break; }
+        }
+        if (!ignored) { return UNKNOWN_FIELD; }
+      }
+      continue;
+    }
+    const std::size_t field_index = key_fields[key_index];
+    if (seen_field[field_index]) { continue; }
+    seen_field[field_index] = true;
+    SIMDJSON_TRY(deserialize_field_at(field_index, json_field.value(), out));
+  }
+  return handle_missing_fields(seen_field, out);
+}
+
+template <typename T>
+consteval bool is_transparent() {
+  return simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag);
+}
+
+} // namespace key_selector_reflection_detail
+
+// Deserialize a reflected struct. By default this builds a compile-time
+// key_selector from the struct's members and walks each object once with
+// object::for_each (perfect-hash key matching), instead of one obj[key] lookup
+// per member. Other paths are used instead:
+//   - globally, the ordered per-member path when defining
+//     -DSIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION=1;
+//   - automatically and per-type, a scan of the object comparing unescaped keys
+//     when the struct's keys do not fit the key_selector limits (see
+//     keys_fit_selector) or cannot be compared raw (see keys_need_unescaping),
+//     so that long member names and the like keep compiling rather than
+//     tripping a static_assert.
+// Structs annotated with deny_unknown_fields always use the strict scan, and
+// structs annotated with transparent are deserialized as their single member.
+//
+// noexcept(false): a member's tag_invoke may throw and the exception must reach
+// the caller of get<T>().
 template <typename T, typename ValT>
   requires(user_defined_type<T> && std::is_class_v<T>)
-error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
+  if constexpr (key_selector_reflection_detail::is_transparent<T>()) {
+    constexpr auto mem = simdjson::detail::transparent_member(^^T);
+    if constexpr (std::is_same_v<std::remove_cvref_t<ValT>, lasx::ondemand::object>) {
+      // We were handed an object: only a structure can be deserialized from it.
+      if constexpr (user_defined_type<typename [: std::meta::type_of(mem) :]>) {
+        return tag_invoke(deserialize_tag{}, val, out.[:mem:]);
+      } else {
+        return INCORRECT_TYPE;
+      }
+    } else {
+      return key_selector_reflection_detail::deserialize_member<mem>(val, out.[:mem:]);
+    }
+  } else {
+  static_assert(key_selector_reflection_detail::accepted_keys_are_distinct<T>(),
+                "two members of this structure accept the same JSON key (check rename, alias, "
+                "rename_all and flatten)");
   lasx::ondemand::object obj;
   if constexpr (std::is_same_v<std::remove_cvref_t<ValT>, lasx::ondemand::object>) {
     obj = val;
   } else {
     SIMDJSON_TRY(val.get_object().get(obj));
   }
-  template for (constexpr auto mem : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
-    if constexpr (!std::meta::is_const(mem) && std::meta::is_public(mem)) {
-      constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
-      if constexpr (concepts::optional_type<decltype(out.[:mem:])>) {
-        // for optional members, it's ok if the key is missing
-        auto error = obj[key].get(out.[:mem:]);
-        if (error && error != NO_SUCH_FIELD) {
-          if(error == NO_SUCH_FIELD) {
-            out.[:mem:].reset();
-            continue;
-          }
-          return error;
-        }
-      } else {
-        // for non-optional members, the key must be present
-        SIMDJSON_TRY(obj[key].get(out.[:mem:]));
+  if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::deny_unknown_fields_tag)) {
+    return key_selector_reflection_detail::deserialize_struct_scan<true>(obj, out);
+  } else {
+#if defined(SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION) && SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION
+  // Opt-out build: use the ordered per-member path, unless obj[key] cannot
+  // match T's keys.
+  if constexpr (key_selector_reflection_detail::keys_need_unescaping<T>()) {
+    return key_selector_reflection_detail::deserialize_struct_scan<false>(obj, out);
+  } else {
+    return key_selector_reflection_detail::deserialize_struct_ordered(obj, out);
+  }
+#else
+  if constexpr (key_selector_reflection_detail::eligible_field_count<T>() == 0) {
+    // No fields to deserialize: an empty key_selector cannot be built, so just
+    // validate that the input is an object (done above) and succeed. Mirrors the
+    // ordered per-member path, which iterates over zero members.
+    (void)out;
+    (void)obj;
+    return SUCCESS;
+  } else if constexpr (!key_selector_reflection_detail::keys_fit_selector<T>()) {
+    // Automatic fallback: T's accepted keys do not fit the key_selector limits
+    // (e.g. a member name longer than 63 characters, or a key with a double
+    // quote), so building a selector would be a compile error. Scan the object
+    // instead, so the default never breaks a struct that the opt-out path would
+    // accept.
+    return key_selector_reflection_detail::deserialize_struct_scan<false>(obj, out);
+  } else {
+  using selector = key_selector_reflection_detail::selector_for<T>;
+  if constexpr (key_selector_reflection_detail::all_eligible_fields_required<T>()
+                && !key_selector_reflection_detail::has_aliases<T>()) {
+    // Fast path: every member is required and has a single key. A single
+    // for_each pass parses each matched field; the returned match count then
+    // tells us whether every member was present (matched_count ==
+    // selector::size()) without a per-member "seen" array. A value-parse error
+    // (e.g. a type mismatch) is propagated by for_each.
+    auto walk = obj.template for_each<selector>(
+        [&](std::size_t matched_index, lasx::ondemand::value field_value) -> error_code {
+      std::size_t counter = 0;
+      template for (constexpr auto path : std::define_static_array(key_selector_reflection_detail::eligible_fields(^^T))) {
+        using field = [: path :];
+        if (matched_index == counter) { return key_selector_reflection_detail::deserialize_member<field::leaf>(field_value, field::get(out)); }
+        ++counter;
       }
-    }
-  };
-  return simdjson::SUCCESS;
-}
-
-// Support for enum deserialization - deserialize from string representation using expand approach from P2996R12
+      return SUCCESS;
+    });
+    if (walk.error) { return walk.error; }
+    // A missing required member shows up as a short match count and is reported as
+    // NO_SUCH_FIELD, mirroring the ordered obj[key] path.
+    if (walk.matched_count != selector::size()) { return NO_SUCH_FIELD; }
+    return SUCCESS;
+  } else {
+    static constexpr auto key_fields = std::define_static_array(key_selector_reflection_detail::accepted_key_fields(^^T));
+    std::array<bool, key_selector_reflection_detail::eligible_field_count<T>()> seen_field{};
+    // Single pass over the object: each field whose key matches a member (or one
+    // of its aliases) yields its selector index, which we map back to the
+    // corresponding member. The first key seen for a member wins. The callback
+    // returns an error_code so that a value-parse error (e.g. a type mismatch on
+    // a matched field) is propagated by for_each instead of being silently dropped.
+    error_code walk_error = obj.template for_each<selector>(
+        [&](std::size_t matched_index, lasx::ondemand::value field_value) -> error_code {
+      const std::size_t field_index = key_fields[matched_index];
+      if (seen_field[field_index]) { return SUCCESS; }
+      seen_field[field_index] = true;
+      return key_selector_reflection_detail::deserialize_field_at(field_index, field_value, out);
+    });
+    if (walk_error) { return walk_error; }
+    // Required members must be present: a missing one is reported as
+    // NO_SUCH_FIELD, mirroring the ordered obj[key] path. Optional and defaulted
+    // members may be absent.
+    return key_selector_reflection_detail::handle_missing_fields(seen_field, out);
+  }
+  }
+#endif // SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION
+  }
+  }
+}
+
+// Support for enum deserialization - deserialize from string representation using
+// expand approach from P2996R12. The accepted strings are the enumerator's JSON key
+// (see rename and rename_all) and its aliases.
 template <typename T, typename ValT>
   requires(std::is_enum_v<T>)
 error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
 #if SIMDJSON_STATIC_REFLECTION
   std::string_view str;
   SIMDJSON_TRY(val.get_string().get(str));
-  constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
+  static constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
   template for (constexpr auto enum_val : enumerators) {
-    if (str == std::meta::identifier_of(enum_val)) {
-      out = [:enum_val:];
-      return SUCCESS;
+    template for (constexpr const char *key : std::define_static_array(key_selector_reflection_detail::accepted_keys_of(enum_val))) {
+      if (str == std::string_view(key)) {
+        out = [:enum_val:];
+        return SUCCESS;
+      }
     }
   };

@@ -165346,33 +212597,25 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {

 template <typename simdjson_value, typename T>
   requires(user_defined_type<std::remove_cvref_t<T>>)
-error_code tag_invoke(deserialize_tag, simdjson_value &val, std::unique_ptr<T> &out) noexcept {
-  if (!out) {
-    out = std::make_unique<T>();
-    if (!out) {
-      return MEMALLOC;
-    }
-  }
-  if (auto err = val.get(*out)) {
-    out.reset();
-    return err;
+error_code tag_invoke(deserialize_tag, simdjson_value &val, std::unique_ptr<T> &out) noexcept(false) {
+  std::unique_ptr<T> ptr(new (std::nothrow) T());
+  if (!ptr) {
+    return MEMALLOC;
   }
+  SIMDJSON_TRY(val.get(*ptr));
+  out = std::move(ptr);
   return SUCCESS;
 }

 template <typename simdjson_value, typename T>
   requires(user_defined_type<std::remove_cvref_t<T>>)
-error_code tag_invoke(deserialize_tag, simdjson_value &val, std::shared_ptr<T> &out) noexcept {
-  if (!out) {
-    out = std::make_shared<T>();
-    if (!out) {
-      return MEMALLOC;
-    }
-  }
-  if (auto err = val.get(*out)) {
-    out.reset();
-    return err;
+error_code tag_invoke(deserialize_tag, simdjson_value &val, std::shared_ptr<T> &out) noexcept(false) {
+  std::shared_ptr<T> ptr(new (std::nothrow) T());
+  if (!ptr) {
+    return MEMALLOC;
   }
+  SIMDJSON_TRY(val.get(*ptr));
+  out = std::move(ptr);
   return SUCCESS;
 }

@@ -165684,9 +212927,17 @@ simdjson_inline simdjson_result<array> array::started(value_iterator &iter) noex
   return array(iter);
 }

-simdjson_inline simdjson_result<array_iterator> array::begin() noexcept {
+simdjson_inline simdjson_result<array_iterator> array::begin() & noexcept {
 #if SIMDJSON_DEVELOPMENT_CHECKS
-  if (!iter.is_at_iterator_start()) { return OUT_OF_ORDER_ITERATION; }
+  if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
+  return array_iterator(iter, this);
+#endif
+  return array_iterator(iter);
+}
+simdjson_inline simdjson_result<array_iterator> array::begin() && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  // The array is a temporary that the iterator may outlive: do not lock it.
+  if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
 #endif
   return array_iterator(iter);
 }
@@ -165713,6 +212964,9 @@ simdjson_inline simdjson_result<std::string_view> array::raw_json() noexcept {
 SIMDJSON_PUSH_DISABLE_WARNINGS
 SIMDJSON_DISABLE_STRICT_OVERFLOW_WARNING
 simdjson_inline simdjson_result<size_t> array::count_elements() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   size_t count{0};
   // Important: we do not consume any of the values.
   for(simdjson_unused auto v : *this) { count++; }
@@ -165726,6 +212980,9 @@ simdjson_inline simdjson_result<size_t> array::count_elements() & noexcept {
 SIMDJSON_POP_DISABLE_WARNINGS

 simdjson_inline simdjson_result<bool> array::is_empty() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   bool is_not_empty;
   auto error = iter.reset_array().get(is_not_empty);
   if(error) { return error; }
@@ -165733,31 +212990,30 @@ simdjson_inline simdjson_result<bool> array::is_empty() & noexcept {
 }

 inline simdjson_result<bool> array::reset() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   return iter.reset_array();
 }

+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline void array::set_locked(bool _locked) noexcept {
+  locked = _locked;
+}
+#endif
+
 inline simdjson_result<value> array::at_pointer(std::string_view json_pointer) noexcept {
-  if (json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
+  // An empty pointer has no json_pointer[0]: with a default-constructed
+  // std::string_view, reading it dereferences a null pointer.
+  if (json_pointer.empty() || json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
   json_pointer = json_pointer.substr(1);
   // - means "the append position" or "the element after the end of the array"
   // We don't support this, because we're returning a real element, not a position.
   if (json_pointer == "-") { return INDEX_OUT_OF_BOUNDS; }

-  // Read the array index
   size_t array_index = 0;
   size_t i;
-  for (i = 0; i < json_pointer.length() && json_pointer[i] != '/'; i++) {
-    uint8_t digit = uint8_t(json_pointer[i] - '0');
-    // Check for non-digit in array index. If it's there, we're trying to get a field in an object
-    if (digit > 9) { return INCORRECT_TYPE; }
-    array_index = array_index*10 + digit;
-  }
-
-  // 0 followed by other digits is invalid
-  if (i > 1 && json_pointer[0] == '0') { return INVALID_JSON_POINTER; } // "JSON pointer array index has other characters after 0"
-
-  // Empty string is invalid; so is a "/" with no digits before it
-  if (i == 0) { return INVALID_JSON_POINTER; } // "Empty string in JSON pointer array index"
+  SIMDJSON_TRY(internal::parse_json_pointer_array_index(json_pointer, array_index, i));
   // Get the child
   auto child = at(array_index);
   // If there is an error, it ends here
@@ -165831,6 +213087,9 @@ inline error_code array::for_each_at_path_with_wildcard(std::string_view json_pa
 }

 simdjson_inline simdjson_result<value> array::at(size_t index) noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   size_t i = 0;
   for (auto value : *this) {
     if (i == index) { return value; }
@@ -165860,10 +213119,14 @@ simdjson_inline simdjson_result<lasx::ondemand::array>::simdjson_result(
 {
 }

-simdjson_inline simdjson_result<lasx::ondemand::array_iterator> simdjson_result<lasx::ondemand::array>::begin() noexcept {
+simdjson_inline simdjson_result<lasx::ondemand::array_iterator> simdjson_result<lasx::ondemand::array>::begin() & noexcept {
   if (error()) { return error(); }
   return first.begin();
 }
+simdjson_inline simdjson_result<lasx::ondemand::array_iterator> simdjson_result<lasx::ondemand::array>::begin() && noexcept {
+  if (error()) { return error(); }
+  return std::move(first).begin();
+}
 simdjson_inline simdjson_result<lasx::ondemand::array_iterator> simdjson_result<lasx::ondemand::array>::end() noexcept {
   if (error()) { return error(); }
   return first.end();
@@ -165926,6 +213189,59 @@ simdjson_inline array_iterator::array_iterator(const value_iterator &_iter) noex
   : iter{_iter}
 {}

+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline array_iterator::array_iterator(const value_iterator &_iter, array* _parent) noexcept
+  : parent{_parent}, iter{_iter}
+{
+  if (parent) parent->set_locked(true);
+}
+
+simdjson_inline array_iterator::~array_iterator() noexcept
+{
+  if (parent) parent->set_locked(false);
+}
+
+simdjson_inline array_iterator::array_iterator(array_iterator&& other) noexcept
+  : has_been_referenced{other.has_been_referenced},
+    parent{other.parent},
+    iter{std::move(other.iter)}
+{
+  other.parent = nullptr;
+}
+
+simdjson_inline array_iterator& array_iterator::operator=(array_iterator&& other) noexcept {
+  if (this != &other)
+  {
+    if (parent)
+      parent->set_locked(false);
+    has_been_referenced = other.has_been_referenced;
+    parent = other.parent;
+    iter = std::move(other.iter);
+
+    other.parent = nullptr;
+  }
+  return *this;
+}
+
+simdjson_inline array_iterator::array_iterator(const array_iterator& other) noexcept
+  : has_been_referenced{other.has_been_referenced},
+    parent{nullptr},
+    iter{other.iter}
+{}
+
+simdjson_inline array_iterator& array_iterator::operator=(const array_iterator& other) noexcept {
+  if (this != &other)
+  {
+    if (parent)
+      parent->set_locked(false);
+    has_been_referenced = other.has_been_referenced;
+    parent = nullptr;
+    iter = other.iter;
+  }
+  return *this;
+}
+#endif
+
 simdjson_inline simdjson_result<value> array_iterator::operator*() noexcept {
 #if SIMDJSON_DEVELOPMENT_CHECKS
    SIMDJSON_ASSUME(!has_been_referenced);
@@ -166021,6 +213337,41 @@ namespace simdjson {
 namespace lasx {
 namespace ondemand {

+/**
+ * @private Narrows the result of get_uint64() to a smaller unsigned integer type T.
+ * @returns NUMBER_OUT_OF_RANGE if the value does not fit in a T.
+ */
+template <typename T>
+simdjson_inline simdjson_result<T> narrow_integer(simdjson_result<uint64_t> wide) noexcept {
+  uint64_t result;
+  SIMDJSON_TRY(std::move(wide).get(result));
+  if (result > static_cast<uint64_t>((std::numeric_limits<T>::max)())) { return NUMBER_OUT_OF_RANGE; }
+  return static_cast<T>(result);
+}
+
+/**
+ * @private Narrows the result of get_int64() to a smaller signed integer type T.
+ * @returns NUMBER_OUT_OF_RANGE if the value does not fit in a T.
+ */
+template <typename T>
+simdjson_inline simdjson_result<T> narrow_integer(simdjson_result<int64_t> wide) noexcept {
+  int64_t result;
+  SIMDJSON_TRY(std::move(wide).get(result));
+  if (result > static_cast<int64_t>((std::numeric_limits<T>::max)()) || result < static_cast<int64_t>((std::numeric_limits<T>::min)())) { return NUMBER_OUT_OF_RANGE; }
+  return static_cast<T>(result);
+}
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+// get_float32() parses with get_float(), so float must be binary32.
+static_assert(std::numeric_limits<float>::is_iec559 && std::numeric_limits<float>::digits == 24,
+              "get_float32() requires float to be IEEE binary32");
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+// get_float64() parses with get_double(), so double must be binary64.
+static_assert(std::numeric_limits<double>::is_iec559 && std::numeric_limits<double>::digits == 53,
+              "get_float64() requires double to be IEEE binary64");
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
+
 simdjson_inline value::value(const value_iterator &_iter) noexcept
   : iter{_iter}
 {
@@ -166052,6 +213403,13 @@ simdjson_inline simdjson_result<raw_json_string> value::get_raw_json_string() no
 simdjson_inline simdjson_result<std::string_view> value::get_string(bool allow_replacement) noexcept {
   return iter.get_string(allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> value::get_u8string(bool allow_replacement) noexcept {
+  std::string_view content;
+  SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+  return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code value::get_string(string_type& receiver, bool allow_replacement) noexcept {
   return iter.get_string(receiver, allow_replacement);
@@ -166065,6 +213423,12 @@ simdjson_inline simdjson_result<double> value::get_double() noexcept {
 simdjson_inline simdjson_result<double> value::get_double_in_string() noexcept {
   return iter.get_double_in_string();
 }
+simdjson_inline simdjson_result<float> value::get_float() noexcept {
+  return iter.get_float();
+}
+simdjson_inline simdjson_result<float> value::get_float_in_string() noexcept {
+  return iter.get_float_in_string();
+}
 simdjson_inline simdjson_result<uint64_t> value::get_uint64() noexcept {
   return iter.get_uint64();
 }
@@ -166078,17 +213442,37 @@ simdjson_inline simdjson_result<int64_t> value::get_int64_in_string() noexcept {
   return iter.get_int64_in_string();
 }
 simdjson_inline simdjson_result<uint32_t> value::get_uint32() noexcept {
-  uint64_t result;
-  SIMDJSON_TRY(get_uint64().get(result));
-  if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<uint32_t>(result);
+  return narrow_integer<uint32_t>(get_uint64());
 }
 simdjson_inline simdjson_result<int32_t> value::get_int32() noexcept {
-  int64_t result;
-  SIMDJSON_TRY(get_int64().get(result));
-  if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<int32_t>(result);
+  return narrow_integer<int32_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint16_t> value::get_uint16() noexcept {
+  return narrow_integer<uint16_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int16_t> value::get_int16() noexcept {
+  return narrow_integer<int16_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint8_t> value::get_uint8() noexcept {
+  return narrow_integer<uint8_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int8_t> value::get_int8() noexcept {
+  return narrow_integer<int8_t>(get_int64());
 }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> value::get_float32() noexcept {
+  float result;
+  SIMDJSON_TRY(get_float().get(result));
+  return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> value::get_float64() noexcept {
+  double result;
+  SIMDJSON_TRY(get_double().get(result));
+  return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<bool> value::get_bool() noexcept {
   return iter.get_bool();
 }
@@ -166100,12 +213484,26 @@ template<> simdjson_inline simdjson_result<array> value::get() noexcept { return
 template<> simdjson_inline simdjson_result<object> value::get() noexcept { return get_object(); }
 template<> simdjson_inline simdjson_result<raw_json_string> value::get() noexcept { return get_raw_json_string(); }
 template<> simdjson_inline simdjson_result<std::string_view> value::get() noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> value::get() noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_inline simdjson_result<number> value::get() noexcept { return get_number(); }
 template<> simdjson_inline simdjson_result<double> value::get() noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> value::get() noexcept { return get_float(); }
 template<> simdjson_inline simdjson_result<uint64_t> value::get() noexcept { return get_uint64(); }
 template<> simdjson_inline simdjson_result<int64_t> value::get() noexcept { return get_int64(); }
 template<> simdjson_inline simdjson_result<uint32_t> value::get() noexcept { return get_uint32(); }
 template<> simdjson_inline simdjson_result<int32_t> value::get() noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> value::get() noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> value::get() noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> value::get() noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> value::get() noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> value::get() noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> value::get() noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<> simdjson_inline simdjson_result<bool> value::get() noexcept { return get_bool(); }


@@ -166113,12 +213511,26 @@ template<> simdjson_warn_unused simdjson_inline error_code value::get(array& out
 template<> simdjson_warn_unused simdjson_inline error_code value::get(object& out) noexcept { return get_object().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(raw_json_string& out) noexcept { return get_raw_json_string().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(std::string_view& out) noexcept { return get_string(false).get(out); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::u8string_view& out) noexcept { return get_u8string(false).get(out); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_warn_unused simdjson_inline error_code value::get(number& out) noexcept { return get_number().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(double& out) noexcept { return get_double().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(float& out) noexcept { return get_float().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(uint64_t& out) noexcept { return get_uint64().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(int64_t& out) noexcept { return get_int64().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(uint32_t& out) noexcept { return get_uint32().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(int32_t& out) noexcept { return get_int32().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(uint16_t& out) noexcept { return get_uint16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(int16_t& out) noexcept { return get_int16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(uint8_t& out) noexcept { return get_uint8().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(int8_t& out) noexcept { return get_int8().get(out); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::float32_t& out) noexcept { return get_float32().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::float64_t& out) noexcept { return get_float64().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<>  simdjson_warn_unused simdjson_inline error_code value::get(bool& out) noexcept { return get_bool().get(out); }

 #if SIMDJSON_EXCEPTIONS
@@ -166287,6 +213699,9 @@ inline bool is_pointer_well_formed(std::string_view json_pointer) noexcept {
 }

 simdjson_inline simdjson_result<value> value::at_pointer(std::string_view json_pointer) noexcept {
+  // The empty JSON Pointer refers to the whole value (RFC 6901), as in
+  // document::at_pointer.
+  if (json_pointer.empty()) { return value(iter); }
   json_type t;
   SIMDJSON_TRY(type().get(t));
   switch (t)
@@ -166324,6 +213739,10 @@ template <typename Func>
 template <typename Func>
 #endif
 inline error_code value::for_each_at_path_with_wildcard(std::string_view json_path, Func&& callback) noexcept {
+  // Every recursive step of for_each_at_path_with_wildcard goes through this
+  // function, and each one descends one level into the document. A path with
+  // many segments applied to a deeply nested document would otherwise recurse
+  // without bound and overflow the stack.
   if (size_t(iter.depth()) >= iter.json_iter().parser->max_depth()) { return DEPTH_ERROR; }
   json_type t;
   SIMDJSON_TRY(type().get(t));
@@ -166437,10 +213856,46 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<lasx::ondemand::value>:
   if (error()) { return error(); }
   return first.get_int32();
 }
+simdjson_inline simdjson_result<uint16_t> simdjson_result<lasx::ondemand::value>::get_uint16() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<lasx::ondemand::value>::get_int16() noexcept {
+  if (error()) { return error(); }
+  return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<lasx::ondemand::value>::get_uint8() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<lasx::ondemand::value>::get_int8() noexcept {
+  if (error()) { return error(); }
+  return first.get_int8();
+}
 simdjson_inline simdjson_result<double> simdjson_result<lasx::ondemand::value>::get_double() noexcept {
   if (error()) { return error(); }
   return first.get_double();
 }
+simdjson_inline simdjson_result<float> simdjson_result<lasx::ondemand::value>::get_float() noexcept {
+  if (error()) { return error(); }
+  return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<lasx::ondemand::value>::get_float_in_string() noexcept {
+  if (error()) { return error(); }
+  return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<lasx::ondemand::value>::get_float32() noexcept {
+  if (error()) { return error(); }
+  return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<lasx::ondemand::value>::get_float64() noexcept {
+  if (error()) { return error(); }
+  return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<double> simdjson_result<lasx::ondemand::value>::get_double_in_string() noexcept {
   if (error()) { return error(); }
   return first.get_double_in_string();
@@ -166449,6 +213904,12 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<lasx::ondemand
   if (error()) { return error(); }
   return first.get_string(allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<lasx::ondemand::value>::get_u8string(bool allow_replacement) noexcept {
+  if (error()) { return error(); }
+  return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_inline error_code simdjson_result<lasx::ondemand::value>::get_string(string_type& receiver, bool allow_replacement) noexcept {
   if (error()) { return error(); }
@@ -166477,11 +213938,23 @@ template<> simdjson_inline error_code simdjson_result<lasx::ondemand::value>::ge
   return SUCCESS;
 }

-template<typename T> simdjson_inline simdjson_result<T> simdjson_result<lasx::ondemand::value>::get() noexcept {
+template<typename T> simdjson_inline simdjson_result<T> simdjson_result<lasx::ondemand::value>::get()
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lasx::ondemand::value>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>();
 }
-template<typename T> simdjson_inline error_code simdjson_result<lasx::ondemand::value>::get(T &out) noexcept {
+template<typename T> simdjson_inline error_code simdjson_result<lasx::ondemand::value>::get(T &out)
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lasx::ondemand::value>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>(out);
 }
@@ -166751,16 +214224,22 @@ simdjson_inline simdjson_result<int64_t> document::get_int64_in_string() noexcep
   return get_root_value_iterator().get_root_int64_in_string(true);
 }
 simdjson_inline simdjson_result<uint32_t> document::get_uint32() noexcept {
-  uint64_t result;
-  SIMDJSON_TRY(get_uint64().get(result));
-  if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<uint32_t>(result);
+  return narrow_integer<uint32_t>(get_uint64());
 }
 simdjson_inline simdjson_result<int32_t> document::get_int32() noexcept {
-  int64_t result;
-  SIMDJSON_TRY(get_int64().get(result));
-  if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<int32_t>(result);
+  return narrow_integer<int32_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint16_t> document::get_uint16() noexcept {
+  return narrow_integer<uint16_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int16_t> document::get_int16() noexcept {
+  return narrow_integer<int16_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint8_t> document::get_uint8() noexcept {
+  return narrow_integer<uint8_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int8_t> document::get_int8() noexcept {
+  return narrow_integer<int8_t>(get_int64());
 }
 simdjson_inline simdjson_result<double> document::get_double() noexcept {
   return get_root_value_iterator().get_root_double(true);
@@ -166768,9 +214247,36 @@ simdjson_inline simdjson_result<double> document::get_double() noexcept {
 simdjson_inline simdjson_result<double> document::get_double_in_string() noexcept {
   return get_root_value_iterator().get_root_double_in_string(true);
 }
+simdjson_inline simdjson_result<float> document::get_float() noexcept {
+  return get_root_value_iterator().get_root_float(true);
+}
+simdjson_inline simdjson_result<float> document::get_float_in_string() noexcept {
+  return get_root_value_iterator().get_root_float_in_string(true);
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> document::get_float32() noexcept {
+  float result;
+  SIMDJSON_TRY(get_float().get(result));
+  return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> document::get_float64() noexcept {
+  double result;
+  SIMDJSON_TRY(get_double().get(result));
+  return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<std::string_view> document::get_string(bool allow_replacement) noexcept {
   return get_root_value_iterator().get_root_string(true, allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> document::get_u8string(bool allow_replacement) noexcept {
+  std::string_view content;
+  SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+  return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code document::get_string(string_type& receiver, bool allow_replacement) noexcept {
   return get_root_value_iterator().get_root_string(receiver, true, allow_replacement);
@@ -166792,11 +214298,25 @@ template<> simdjson_inline simdjson_result<array> document::get() & noexcept { r
 template<> simdjson_inline simdjson_result<object> document::get() & noexcept { return get_object(); }
 template<> simdjson_inline simdjson_result<raw_json_string> document::get() & noexcept { return get_raw_json_string(); }
 template<> simdjson_inline simdjson_result<std::string_view> document::get() & noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> document::get() & noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_inline simdjson_result<double> document::get() & noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> document::get() & noexcept { return get_float(); }
 template<> simdjson_inline simdjson_result<uint64_t> document::get() & noexcept { return get_uint64(); }
 template<> simdjson_inline simdjson_result<int64_t> document::get() & noexcept { return get_int64(); }
 template<> simdjson_inline simdjson_result<uint32_t> document::get() & noexcept { return get_uint32(); }
 template<> simdjson_inline simdjson_result<int32_t> document::get() & noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> document::get() & noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> document::get() & noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> document::get() & noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> document::get() & noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> document::get() & noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> document::get() & noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<> simdjson_inline simdjson_result<bool> document::get() & noexcept { return get_bool(); }
 template<> simdjson_inline simdjson_result<value> document::get() & noexcept { return get_value(); }

@@ -166804,17 +214324,35 @@ template<> simdjson_warn_unused simdjson_inline error_code document::get(array&
 template<> simdjson_warn_unused simdjson_inline error_code document::get(object& out) & noexcept { return get_object().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(raw_json_string& out) & noexcept { return get_raw_json_string().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(std::string_view& out) & noexcept { return get_string(false).get(out); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::u8string_view& out) & noexcept { return get_u8string(false).get(out); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_warn_unused simdjson_inline error_code document::get(double& out) & noexcept { return get_double().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(float& out) & noexcept { return get_float().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(uint64_t& out) & noexcept { return get_uint64().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(int64_t& out) & noexcept { return get_int64().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(uint32_t& out) & noexcept { return get_uint32().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(int32_t& out) & noexcept { return get_int32().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(uint16_t& out) & noexcept { return get_uint16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(int16_t& out) & noexcept { return get_int16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(uint8_t& out) & noexcept { return get_uint8().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(int8_t& out) & noexcept { return get_int8().get(out); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::float32_t& out) & noexcept { return get_float32().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::float64_t& out) & noexcept { return get_float64().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<> simdjson_warn_unused simdjson_inline error_code document::get(bool& out) & noexcept { return get_bool().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(value& out) & noexcept { return get_value().get(out); }

 template<> simdjson_deprecated simdjson_inline simdjson_result<raw_json_string> document::get() && noexcept { return get_raw_json_string(); }
 template<> simdjson_deprecated simdjson_inline simdjson_result<std::string_view> document::get() && noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_deprecated simdjson_inline simdjson_result<std::u8string_view> document::get() && noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_deprecated simdjson_inline simdjson_result<double> document::get() && noexcept { return std::forward<document>(*this).get_double(); }
+template<> simdjson_deprecated simdjson_inline simdjson_result<float> document::get() && noexcept { return std::forward<document>(*this).get_float(); }
 template<> simdjson_deprecated simdjson_inline simdjson_result<uint64_t> document::get() && noexcept { return std::forward<document>(*this).get_uint64(); }
 template<> simdjson_deprecated simdjson_inline simdjson_result<int64_t> document::get() && noexcept { return std::forward<document>(*this).get_int64(); }
 template<> simdjson_deprecated simdjson_inline simdjson_result<bool> document::get() && noexcept { return std::forward<document>(*this).get_bool(); }
@@ -167153,6 +214691,22 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<lasx::ondemand::documen
   if (error()) { return error(); }
   return first.get_int32();
 }
+simdjson_inline simdjson_result<uint16_t> simdjson_result<lasx::ondemand::document>::get_uint16() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<lasx::ondemand::document>::get_int16() noexcept {
+  if (error()) { return error(); }
+  return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<lasx::ondemand::document>::get_uint8() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<lasx::ondemand::document>::get_int8() noexcept {
+  if (error()) { return error(); }
+  return first.get_int8();
+}
 simdjson_inline simdjson_result<double> simdjson_result<lasx::ondemand::document>::get_double() noexcept {
   if (error()) { return error(); }
   return first.get_double();
@@ -167161,10 +214715,36 @@ simdjson_inline simdjson_result<double> simdjson_result<lasx::ondemand::document
   if (error()) { return error(); }
   return first.get_double_in_string();
 }
+simdjson_inline simdjson_result<float> simdjson_result<lasx::ondemand::document>::get_float() noexcept {
+  if (error()) { return error(); }
+  return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<lasx::ondemand::document>::get_float_in_string() noexcept {
+  if (error()) { return error(); }
+  return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<lasx::ondemand::document>::get_float32() noexcept {
+  if (error()) { return error(); }
+  return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<lasx::ondemand::document>::get_float64() noexcept {
+  if (error()) { return error(); }
+  return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<std::string_view> simdjson_result<lasx::ondemand::document>::get_string(bool allow_replacement) noexcept {
   if (error()) { return error(); }
   return first.get_string(allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<lasx::ondemand::document>::get_u8string(bool allow_replacement) noexcept {
+  if (error()) { return error(); }
+  return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code simdjson_result<lasx::ondemand::document>::get_string(string_type& receiver, bool allow_replacement) noexcept {
   if (error()) { return error(); }
@@ -167192,22 +214772,46 @@ simdjson_inline simdjson_result<bool> simdjson_result<lasx::ondemand::document>:
 }

 template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<lasx::ondemand::document>::get() & noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<lasx::ondemand::document>::get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lasx::ondemand::document>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>();
 }
 template<typename T>
-simdjson_deprecated simdjson_inline simdjson_result<T> simdjson_result<lasx::ondemand::document>::get() && noexcept {
+simdjson_deprecated simdjson_inline simdjson_result<T> simdjson_result<lasx::ondemand::document>::get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lasx::ondemand::document>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return std::forward<lasx::ondemand::document>(first).get<T>();
 }
 template<typename T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<lasx::ondemand::document>::get(T &out) & noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<lasx::ondemand::document>::get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lasx::ondemand::document>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>(out);
 }
 template<typename T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<lasx::ondemand::document>::get(T &out) && noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<lasx::ondemand::document>::get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lasx::ondemand::document>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return std::forward<lasx::ondemand::document>(first).get<T>(out);
 }
@@ -167276,27 +214880,27 @@ simdjson_inline simdjson_result<lasx::ondemand::document>::operator lasx::ondema
 }
 simdjson_inline simdjson_result<lasx::ondemand::document>::operator uint64_t() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_uint64();
 }
 simdjson_inline simdjson_result<lasx::ondemand::document>::operator int64_t() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_int64();
 }
 simdjson_inline simdjson_result<lasx::ondemand::document>::operator double() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_double();
 }
 simdjson_inline simdjson_result<lasx::ondemand::document>::operator std::string_view() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_string();
 }
 simdjson_inline simdjson_result<lasx::ondemand::document>::operator lasx::ondemand::raw_json_string() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_raw_json_string();
 }
 simdjson_inline simdjson_result<lasx::ondemand::document>::operator bool() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_bool();
 }
 simdjson_inline simdjson_result<lasx::ondemand::document>::operator lasx::ondemand::value() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
@@ -167386,21 +214990,38 @@ simdjson_inline simdjson_result<uint64_t> document_reference::get_uint64() noexc
 simdjson_inline simdjson_result<uint64_t> document_reference::get_uint64_in_string() noexcept { return doc->get_root_value_iterator().get_root_uint64_in_string(false); }
 simdjson_inline simdjson_result<int64_t> document_reference::get_int64() noexcept { return doc->get_root_value_iterator().get_root_int64(false); }
 simdjson_inline simdjson_result<int64_t> document_reference::get_int64_in_string() noexcept { return doc->get_root_value_iterator().get_root_int64_in_string(false); }
-simdjson_inline simdjson_result<uint32_t> document_reference::get_uint32() noexcept {
-  uint64_t result;
-  SIMDJSON_TRY(doc->get_root_value_iterator().get_root_uint64(false).get(result));
-  if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<uint32_t>(result);
-}
-simdjson_inline simdjson_result<int32_t> document_reference::get_int32() noexcept {
-  int64_t result;
-  SIMDJSON_TRY(doc->get_root_value_iterator().get_root_int64(false).get(result));
-  if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<int32_t>(result);
-}
+simdjson_inline simdjson_result<uint32_t> document_reference::get_uint32() noexcept { return narrow_integer<uint32_t>(get_uint64()); }
+simdjson_inline simdjson_result<int32_t> document_reference::get_int32() noexcept { return narrow_integer<int32_t>(get_int64()); }
+simdjson_inline simdjson_result<uint16_t> document_reference::get_uint16() noexcept { return narrow_integer<uint16_t>(get_uint64()); }
+simdjson_inline simdjson_result<int16_t> document_reference::get_int16() noexcept { return narrow_integer<int16_t>(get_int64()); }
+simdjson_inline simdjson_result<uint8_t> document_reference::get_uint8() noexcept { return narrow_integer<uint8_t>(get_uint64()); }
+simdjson_inline simdjson_result<int8_t> document_reference::get_int8() noexcept { return narrow_integer<int8_t>(get_int64()); }
 simdjson_inline simdjson_result<double> document_reference::get_double() noexcept { return doc->get_root_value_iterator().get_root_double(false); }
-simdjson_inline simdjson_result<double> document_reference::get_double_in_string() noexcept { return doc->get_root_value_iterator().get_root_double(false); }
+simdjson_inline simdjson_result<double> document_reference::get_double_in_string() noexcept { return doc->get_root_value_iterator().get_root_double_in_string(false); }
+simdjson_inline simdjson_result<float> document_reference::get_float() noexcept { return doc->get_root_value_iterator().get_root_float(false); }
+simdjson_inline simdjson_result<float> document_reference::get_float_in_string() noexcept { return doc->get_root_value_iterator().get_root_float_in_string(false); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> document_reference::get_float32() noexcept {
+  float result;
+  SIMDJSON_TRY(get_float().get(result));
+  return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> document_reference::get_float64() noexcept {
+  double result;
+  SIMDJSON_TRY(get_double().get(result));
+  return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<std::string_view> document_reference::get_string(bool allow_replacement) noexcept { return doc->get_root_value_iterator().get_root_string(false, allow_replacement); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> document_reference::get_u8string(bool allow_replacement) noexcept {
+  std::string_view content;
+  SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+  return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code document_reference::get_string(string_type& receiver, bool allow_replacement) noexcept { return doc->get_root_value_iterator().get_root_string(receiver, false, allow_replacement); }
 simdjson_inline simdjson_result<std::string_view> document_reference::get_wobbly_string() noexcept { return doc->get_root_value_iterator().get_root_wobbly_string(false); }
@@ -167412,11 +215033,25 @@ template<> simdjson_inline simdjson_result<array> document_reference::get() & no
 template<> simdjson_inline simdjson_result<object> document_reference::get() & noexcept { return get_object(); }
 template<> simdjson_inline simdjson_result<raw_json_string> document_reference::get() & noexcept { return get_raw_json_string(); }
 template<> simdjson_inline simdjson_result<std::string_view> document_reference::get() & noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> document_reference::get() & noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_inline simdjson_result<double> document_reference::get() & noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> document_reference::get() & noexcept { return get_float(); }
 template<> simdjson_inline simdjson_result<uint64_t> document_reference::get() & noexcept { return get_uint64(); }
 template<> simdjson_inline simdjson_result<int64_t> document_reference::get() & noexcept { return get_int64(); }
 template<> simdjson_inline simdjson_result<uint32_t> document_reference::get() & noexcept { return get_uint32(); }
 template<> simdjson_inline simdjson_result<int32_t> document_reference::get() & noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> document_reference::get() & noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> document_reference::get() & noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> document_reference::get() & noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> document_reference::get() & noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> document_reference::get() & noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> document_reference::get() & noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<> simdjson_inline simdjson_result<bool> document_reference::get() & noexcept { return get_bool(); }
 template<> simdjson_inline simdjson_result<value> document_reference::get() & noexcept { return get_value(); }
 #if SIMDJSON_EXCEPTIONS
@@ -167562,6 +215197,22 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<lasx::ondemand::documen
   if (error()) { return error(); }
   return first.get_int32();
 }
+simdjson_inline simdjson_result<uint16_t> simdjson_result<lasx::ondemand::document_reference>::get_uint16() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<lasx::ondemand::document_reference>::get_int16() noexcept {
+  if (error()) { return error(); }
+  return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<lasx::ondemand::document_reference>::get_uint8() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<lasx::ondemand::document_reference>::get_int8() noexcept {
+  if (error()) { return error(); }
+  return first.get_int8();
+}
 simdjson_inline simdjson_result<double> simdjson_result<lasx::ondemand::document_reference>::get_double() noexcept {
   if (error()) { return error(); }
   return first.get_double();
@@ -167570,10 +215221,36 @@ simdjson_inline simdjson_result<double> simdjson_result<lasx::ondemand::document
   if (error()) { return error(); }
   return first.get_double_in_string();
 }
+simdjson_inline simdjson_result<float> simdjson_result<lasx::ondemand::document_reference>::get_float() noexcept {
+  if (error()) { return error(); }
+  return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<lasx::ondemand::document_reference>::get_float_in_string() noexcept {
+  if (error()) { return error(); }
+  return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<lasx::ondemand::document_reference>::get_float32() noexcept {
+  if (error()) { return error(); }
+  return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<lasx::ondemand::document_reference>::get_float64() noexcept {
+  if (error()) { return error(); }
+  return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<std::string_view> simdjson_result<lasx::ondemand::document_reference>::get_string(bool allow_replacement) noexcept {
   if (error()) { return error(); }
   return first.get_string(allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<lasx::ondemand::document_reference>::get_u8string(bool allow_replacement) noexcept {
+  if (error()) { return error(); }
+  return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code simdjson_result<lasx::ondemand::document_reference>::get_string(string_type& receiver, bool allow_replacement) noexcept {
   if (error()) { return error(); }
@@ -167600,22 +215277,46 @@ simdjson_inline simdjson_result<bool> simdjson_result<lasx::ondemand::document_r
   return first.is_null();
 }
 template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<lasx::ondemand::document_reference>::get() & noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<lasx::ondemand::document_reference>::get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lasx::ondemand::document_reference>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>();
 }
 template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<lasx::ondemand::document_reference>::get() && noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<lasx::ondemand::document_reference>::get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lasx::ondemand::document_reference>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return std::forward<lasx::ondemand::document_reference>(first).get<T>();
 }
 template <class T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<lasx::ondemand::document_reference>::get(T &out) & noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<lasx::ondemand::document_reference>::get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lasx::ondemand::document_reference>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>(out);
 }
 template <class T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<lasx::ondemand::document_reference>::get(T &out) && noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<lasx::ondemand::document_reference>::get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, lasx::ondemand::document_reference>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return std::forward<lasx::ondemand::document_reference>(first).get<T>(out);
 }
@@ -167677,27 +215378,27 @@ simdjson_inline simdjson_result<lasx::ondemand::document_reference>::operator la
 }
 simdjson_inline simdjson_result<lasx::ondemand::document_reference>::operator uint64_t() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_uint64();
 }
 simdjson_inline simdjson_result<lasx::ondemand::document_reference>::operator int64_t() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_int64();
 }
 simdjson_inline simdjson_result<lasx::ondemand::document_reference>::operator double() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_double();
 }
 simdjson_inline simdjson_result<lasx::ondemand::document_reference>::operator std::string_view() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_string();
 }
 simdjson_inline simdjson_result<lasx::ondemand::document_reference>::operator lasx::ondemand::raw_json_string() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_raw_json_string();
 }
 simdjson_inline simdjson_result<lasx::ondemand::document_reference>::operator bool() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_bool();
 }
 simdjson_inline simdjson_result<lasx::ondemand::document_reference>::operator lasx::ondemand::value() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
@@ -167763,6 +215464,7 @@ simdjson_warn_unused simdjson_inline error_code simdjson_result<lasx::ondemand::
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */

 #include <algorithm>
+#include <cstring>
 #include <stdexcept>

 namespace simdjson {
@@ -167849,23 +215551,20 @@ simdjson_inline document_stream::document_stream(
   const uint8_t *_buf,
   size_t _len,
   size_t _batch_size,
-  bool _allow_comma_separated
+  bool _allow_comma_separated,
+  stream_format _format
 ) noexcept
   : parser{&_parser},
     buf{_buf},
     len{_len},
     batch_size{_batch_size <= MINIMAL_BATCH_SIZE ? MINIMAL_BATCH_SIZE : _batch_size},
     allow_comma_separated{_allow_comma_separated},
+    format{_format},
     error{SUCCESS}
     #ifdef SIMDJSON_THREADS_ENABLED
     , use_thread(_parser.threaded) // we need to make a copy because _parser.threaded can change
     #endif
 {
-#ifdef SIMDJSON_THREADS_ENABLED
-  if(worker.get() == nullptr) {
-    error = MEMALLOC;
-  }
-#endif
 }

 simdjson_inline document_stream::document_stream() noexcept
@@ -167874,6 +215573,7 @@ simdjson_inline document_stream::document_stream() noexcept
     len{0},
     batch_size{0},
     allow_comma_separated{false},
+    format{stream_format::whitespace_delimited},
     error{UNINITIALIZED}
     #ifdef SIMDJSON_THREADS_ENABLED
     , use_thread(false)
@@ -167893,6 +215593,9 @@ inline size_t document_stream::size_in_bytes() const noexcept {
 }

 inline size_t document_stream::truncated_bytes() const noexcept {
+  // Stage 1 returns EMPTY on zero-length input before it writes the index
+  // sentinels read below, so they would still hold a previous stream's values.
+  if (len == 0) { return 0; }
   if(error == CAPACITY) { return len - batch_start; }
   return parser->implementation->structural_indexes[parser->implementation->n_structural_indexes] - parser->implementation->structural_indexes[parser->implementation->n_structural_indexes + 1];
 }
@@ -167973,13 +215676,20 @@ inline void document_stream::start() noexcept {
     error = run_stage1(*parser, batch_start);
   }
   if (error) { return; }
-  doc_index = batch_start;
+  // For json_sequence mode, structural_indexes[0] points to the actual JSON value
+  // after the RS delimiter and any following whitespace. For regular mode, it is
+  // the offset from batch_start to the first document in the batch.
+  doc_index = batch_start + parser->implementation->structural_indexes[0];
   doc = document(json_iterator(&buf[batch_start], parser));
   doc.iter._streaming = true;

   #ifdef SIMDJSON_THREADS_ENABLED
   if (use_thread && next_batch_start() < len) {
     // Kick off the first thread on next batch if needed
+    if (worker.get() == nullptr) {
+      worker.reset(new(std::nothrow) stage1_worker());
+      if (worker.get() == nullptr) { error = MEMALLOC; return; }
+    }
     error = stage1_thread_parser.allocate(batch_size);
     if (error) { return; }
     worker->start_thread();
@@ -168054,12 +215764,69 @@ inline void document_stream::next() noexcept {
        */

       if (error) { continue; } // If the error was EMPTY, we may want to load another batch.
-      doc_index = batch_start;
+      doc_index = batch_start + parser->implementation->structural_indexes[0];
     }
   }
 }

+simdjson_inline uint8_t document_stream::document_delimiter() const noexcept {
+  switch (format) {
+    case stream_format::newline_delimited: return '\n';
+    case stream_format::json_sequence: return 0x1E;
+    default: return 0;
+  }
+}
+
+simdjson_inline bool document_stream::skip_to_delimiter(uint8_t delimiter) noexcept {
+  const uint8_t *const base = &buf[batch_start];
+  const token_position pos = doc.iter.position();
+  const token_position end = doc.iter.end_position();
+  if (pos >= end) { return false; }
+  const size_t here = size_t(doc.iter.token.peek(pos) - base);
+  const size_t batch_len =
+      (len - batch_start < batch_size) ? len - batch_start : batch_size;
+  if (here >= batch_len) { return false; }
+  const uint8_t *const found = static_cast<const uint8_t *>(
+      std::memchr(base + here, delimiter, batch_len - here));
+  if (found == nullptr) { return false; }
+
+  const uint32_t boundary = uint32_t(found - base);
+  // The answer is near `pos`: the delimiter ends the current document, while
+  // `end` spans the whole batch. Gallop first so the cost follows the distance
+  // rather than the size of the batch.
+  token_position lo = pos;
+  size_t hop = 1;
+  while (lo + hop < end && lo[hop] < boundary) { lo += hop; hop <<= 1; }
+  token_position hi = (lo + hop < end) ? lo + hop : end;
+  while (lo < hi) {
+    const token_position mid = lo + ((hi - lo) >> 1);
+    if (*mid < boundary) { lo = mid + 1; } else { hi = mid; }
+  }
+  doc.iter.token.set_position(lo);
+  return true;
+}
+
 inline void document_stream::next_document() noexcept {
+  // A delimiter that cannot occur inside a document tells us where the current
+  // one ends, so we can jump there instead of walking every structural. Only
+  // valid while the iterator is still inside the document: a consumed document
+  // already sits on the next one's first token, and skip_child() returns at
+  // once for it.
+  //
+  // The jump does not structure-validate the unread remainder of the current
+  // document: under newline_delimited / json_sequence the next delimiter is
+  // assumed to be the true document boundary. Callers that leave depth() > 0
+  // while violating that contract (e.g. pretty multi-line JSON under
+  // newline_delimited) can mis-align following documents; use
+  // whitespace_delimited if unsure.
+  const uint8_t delimiter = document_delimiter();
+  if (delimiter != 0 && !error && doc.iter.depth() > 0 &&
+      skip_to_delimiter(delimiter)) {
+    doc.iter._depth = 1;
+    doc.iter._string_buf_loc = parser->string_buf.get();
+    doc.iter._root = doc.iter.position();
+    return;
+  }
   // Go to next place where depth=0 (document depth)
   error = doc.iter.skip_child(0);
   if (error) { return; }
@@ -168083,10 +215850,35 @@ inline error_code document_stream::run_stage1(ondemand::parser &p, size_t _batch
   // This code only updates the structural index in the parser, it does not update any json_iterator
   // instance.
   size_t remaining = len - _batch_start;
+  stage1_mode mode;
   if (remaining <= batch_size) {
-    return p.implementation->stage1(&buf[_batch_start], remaining, stage1_mode::streaming_final);
+    // Final batch
+    switch (format) {
+      case stream_format::json_sequence:
+        mode = stage1_mode::json_sequence_final;
+        break;
+      case stream_format::comma_delimited:
+        mode = stage1_mode::comma_delimited_final;
+        break;
+      default:
+        mode = stage1_mode::streaming_final;
+        break;
+    }
+    return p.implementation->stage1(&buf[_batch_start], remaining, mode);
   } else {
-    return p.implementation->stage1(&buf[_batch_start], batch_size, stage1_mode::streaming_partial);
+    // Partial batch
+    switch (format) {
+      case stream_format::json_sequence:
+        mode = stage1_mode::json_sequence_partial;
+        break;
+      case stream_format::comma_delimited:
+        mode = stage1_mode::comma_delimited_partial;
+        break;
+      default:
+        mode = stage1_mode::streaming_partial;
+        break;
+    }
+    return p.implementation->stage1(&buf[_batch_start], batch_size, mode);
   }
 }

@@ -168095,11 +215887,19 @@ simdjson_inline size_t document_stream::iterator::current_index() const noexcept
 }

 simdjson_inline std::string_view document_stream::iterator::source() const noexcept {
-  auto depth = stream->doc.iter.depth();
+  // On error (e.g., CAPACITY), there is no document to walk: return the rest of
+  // the input, as the DOM document_stream does.
+  if (stream->error) {
+    return std::string_view(reinterpret_cast<const char*>(stream->buf) + current_index(), stream->len - current_index());
+  }
+  // Always walk from the root of the document, whatever the current position
+  // of the document iterator: the user may have already consumed part of the
+  // document, so the iterator's current depth must not be used here.
+  depth_t depth = 1;
   auto cur_struct_index = stream->doc.iter._root - stream->parser->implementation->structural_indexes.get();

-  // If at root, process the first token to determine if scalar value
-  if (stream->doc.iter.at_root()) {
+  // Process the first token to determine if scalar value
+  {
     switch (stream->buf[stream->batch_start + stream->parser->implementation->structural_indexes[cur_struct_index]]) {
       case '{': case '[':   // Depth=1 already at start of document
         break;
@@ -168107,14 +215907,47 @@ simdjson_inline std::string_view document_stream::iterator::source() const noexc
         depth--;
         break;
       default:    // Scalar value document
-        // TODO: We could remove trailing whitespaces
         // This returns a string spanning from start of value to the beginning of the next document (excluded)
         {
           auto next_index = stream->batch_start + stream->parser->implementation->structural_indexes[++cur_struct_index];
           // normally the length would be next_index - current_index() - 1, except for the last document
           size_t svlen = next_index - current_index();
           const char *start = reinterpret_cast<const char*>(stream->buf) + current_index();
-          while(svlen > 1 && (std::isspace(start[svlen-1]) || start[svlen-1] == '\0')) {
+          // When the scalar is followed by a truncated document, the structural
+          // indexes of that document were dropped and next_index is the end of
+          // the input, so we bound the scalar by scanning the token itself.
+          size_t token_len = 0;
+          if (*start == '"') {
+            token_len = 1;
+            while (token_len < svlen) {
+              char c = start[token_len++];
+              if (c == '\\') {
+                token_len++;
+              } else if (c == '"') {
+                break;
+              }
+            }
+          } else {
+            while (token_len < svlen) {
+              char c = start[token_len];
+              if (std::isspace(static_cast<unsigned char>(c)) || c == ',' || c == '{' || c == '[' || c == '\0' || static_cast<uint8_t>(c) == 0x1E) {
+                break;
+              }
+              token_len++;
+            }
+          }
+          if (token_len > 0 && token_len < svlen) {
+            svlen = token_len;
+          }
+          // Trim trailing whitespace, NUL, and RS (0x1E). In RFC 7464
+          // json_sequence mode the scanner classifies RS as a scalar
+          // character, so an RS-prefixed scalar document (number / true /
+          // false / null / string) has no closing structural index and the
+          // slice runs all the way up to the next document's RS. RS cannot
+          // legally appear in a JSON value at the source level (control
+          // characters in strings must be escaped as \u001E), so stripping
+          // it is safe in every stream_format.
+          while(svlen > 1 && (std::isspace(static_cast<unsigned char>(start[svlen-1])) || start[svlen-1] == '\0' || static_cast<uint8_t>(start[svlen-1]) == 0x1E || (stream->format == stream_format::comma_delimited && start[svlen-1] == ','))) {
             svlen--;
           }
           return std::string_view(start, svlen);
@@ -168239,11 +216072,19 @@ simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> field::un
   return answer;
 }

+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_warn_unused simdjson_result<std::u8string_view> field::unescaped_u8key(bool allow_replacement) noexcept {
+  std::string_view key;
+  SIMDJSON_TRY( unescaped_key(allow_replacement).get(key) );
+  return internal::as_u8string_view(key);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 template <typename string_type>
 simdjson_inline simdjson_warn_unused error_code field::unescaped_key(string_type& receiver, bool allow_replacement) noexcept {
   std::string_view key;
   SIMDJSON_TRY( unescaped_key(allow_replacement).get(key) );
-  receiver = key;
+  internal::assign_utf8(receiver, key);
   return SUCCESS;
 }

@@ -168265,6 +216106,12 @@ simdjson_inline std::string_view field::escaped_key() const noexcept {
   return std::string_view(reinterpret_cast<const char*>(first.buf), end_quote - first.buf);
 }

+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline std::u8string_view field::escaped_u8key() const noexcept {
+  return internal::as_u8string_view(escaped_key());
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 simdjson_inline value &field::value() & noexcept {
   return second;
 }
@@ -168309,11 +216156,25 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<lasx::ondemand
   return first.escaped_key();
 }

+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<lasx::ondemand::field>::escaped_u8key() noexcept {
+  if (error()) { return error(); }
+  return first.escaped_u8key();
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 simdjson_inline simdjson_result<std::string_view> simdjson_result<lasx::ondemand::field>::unescaped_key(bool allow_replacement) noexcept {
   if (error()) { return error(); }
   return first.unescaped_key(allow_replacement);
 }

+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<lasx::ondemand::field>::unescaped_u8key(bool allow_replacement) noexcept {
+  if (error()) { return error(); }
+  return first.unescaped_u8key(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 template<typename string_type>
 simdjson_warn_unused simdjson_inline error_code simdjson_result<lasx::ondemand::field>::unescaped_key(string_type &receiver, bool allow_replacement) noexcept {
   if (error()) { return error(); }
@@ -168357,6 +216218,9 @@ simdjson_inline json_iterator::json_iterator(json_iterator &&other) noexcept
     _depth{other._depth},
     _root{other._root},
     _streaming{other._streaming}
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+    , _allow_incomplete_json{other._allow_incomplete_json}
+#endif
 {
   other.parser = nullptr;
 }
@@ -168368,6 +216232,9 @@ simdjson_inline json_iterator &json_iterator::operator=(json_iterator &&other) n
   _depth = other._depth;
   _root = other._root;
   _streaming = other._streaming;
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  _allow_incomplete_json = other._allow_incomplete_json;
+#endif
   other.parser = nullptr;
   return *this;
 }
@@ -168394,7 +216261,8 @@ simdjson_inline json_iterator::json_iterator(const uint8_t *buf, ondemand::parse
       _string_buf_loc{parser->string_buf.get()},
       _depth{1},
       _root{parser->implementation->structural_indexes.get()},
-      _streaming{streaming}
+      _streaming{streaming},
+      _allow_incomplete_json{true}

 {
   logger::log_headers();
@@ -168466,7 +216334,8 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::skip_child(depth_
 #endif // SIMDJSON_CHECK_EOF
       break;
     case '"':
-      if(*peek() == ':') {
+      // At the end, peek() would read the sentinel, which points into the padding.
+      if(!at_end() && *peek() == ':') {
         // We are at a key!!!
         // This might happen if you just started an object and you skip it immediately.
         // Performance note: it would be nice to get rid of this check as it is somewhat
@@ -168509,7 +216378,7 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::skip_child(depth_
     }
   }

-  return report_error(TAPE_ERROR, "not enough close braces");
+  return report_error(INCOMPLETE_ARRAY_OR_OBJECT, "not enough close braces");
 }

 SIMDJSON_POP_DISABLE_WARNINGS
@@ -168526,6 +216395,17 @@ simdjson_inline bool json_iterator::streaming() const noexcept {
   return _streaming;
 }

+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+simdjson_inline bool json_iterator::allow_incomplete_json() const noexcept {
+  return _allow_incomplete_json;
+}
+
+simdjson_inline size_t json_iterator::remaining_input_length(const uint8_t *json) const noexcept {
+  const uint8_t *end = token.buf + parser->_document_len;
+  return json < end ? size_t(end - json) : 0;
+}
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+
 simdjson_inline token_position json_iterator::root_position() const noexcept {
   return _root;
 }
@@ -168808,7 +216688,7 @@ inline std::ostream& operator<<(std::ostream& out, json_type type) noexcept {
         case json_type::string: out << "string"; break;
         case json_type::boolean: out << "boolean"; break;
         case json_type::null: out << "null"; break;
-        default: SIMDJSON_UNREACHABLE();
+        case json_type::unknown: out << "unknown"; break;
     }
     return out;
 }
@@ -169147,6 +217027,10 @@ inline void log_line(const json_iterator &iter, token_position index, depth_t de
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_iterator.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value-inl.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/jsonpathutil.h" */
+/* amalgamation skipped (editor-only): #include <utility> */
+/* amalgamation skipped (editor-only): #if SIMDJSON_SUPPORTS_CONCEPTS */
+/* amalgamation skipped (editor-only): #include <tuple> // std::forward_as_tuple/get for the variadic for_each adapter */
+/* amalgamation skipped (editor-only): #endif */
 /* amalgamation skipped (editor-only): #if SIMDJSON_STATIC_REFLECTION */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_string_builder.h"  // for constevalutil::fixed_string */
 /* amalgamation skipped (editor-only): #include <meta> */
@@ -169176,12 +217060,21 @@ simdjson_inline simdjson_result<value> object::find_field_unordered(const std::s
   return value(iter.child());
 }
 simdjson_inline simdjson_result<value> object::operator[](const std::string_view key) & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   return find_field_unordered(key);
 }
 simdjson_inline simdjson_result<value> object::operator[](const std::string_view key) && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   return std::forward<object>(*this).find_field_unordered(key);
 }
 simdjson_inline simdjson_result<value> object::find_field(const std::string_view key) & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   bool has_value;
   SIMDJSON_TRY( iter.find_field_raw(key).get(has_value) );
   if (!has_value) {
@@ -169191,6 +217084,9 @@ simdjson_inline simdjson_result<value> object::find_field(const std::string_view
   return value(iter.child());
 }
 simdjson_inline simdjson_result<value> object::find_field(const std::string_view key) && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   bool has_value;
   SIMDJSON_TRY( iter.find_field_raw(key).get(has_value) );
   if (!has_value) {
@@ -169200,6 +217096,150 @@ simdjson_inline simdjson_result<value> object::find_field(const std::string_view
   return value(iter.child());
 }

+#if SIMDJSON_SUPPORTS_CONCEPTS
+template <typename Selector, typename Func>
+  requires key_selector_type<Selector> &&
+           std::is_invocable_v<Func&, std::size_t, value>
+simdjson_flatten simdjson_inline for_each_result object::for_each(Func&& on_match)
+    noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, value>) {
+  // Single pass driven directly by the value_iterator, mirroring
+  // find_field_unordered_raw + value(iter.child()). Compared to walking via
+  // object_iterator/field, this avoids constructing a simdjson_result<field> and
+  // a field (key + value) for every field -- and the development-check bookkeeping
+  // in object_iterator -- building a value only for the fields that actually match.
+  // We operate on a copy of the iterator, as object::begin() would.
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  // Mirror object::begin(): for_each must start at the beginning of the object,
+  // not from some position left behind by a prior find_field on the same object.
+  if (!iter.is_at_iterator_start()) { return {OUT_OF_ORDER_ITERATION, 0}; }
+#endif
+  value_iterator it = iter;
+  std::size_t matched = 0;
+  // Track which selector indices have already matched, as a compile-time bitset
+  // (one 64-bit word per 64 keys): the callback fires once per key, on its first
+  // occurrence, and we stop as soon as every key has matched.
+  constexpr std::size_t seen_words = (Selector::size() + 63) / 64;
+  std::array<std::uint64_t, seen_words> seen{};
+  while (it.is_open()) {
+    raw_json_string key;
+    error_code error;
+    std::size_t idx;
+    if constexpr (Selector::window.ok) {
+      // A window selector confirms a key from its raw bytes alone (the closing
+      // quote bounds it), so we take the length-free path: field_key (no backward
+      // scan, mirroring the ordered find_field) feeding match_raw(raw_json_string).
+      if ((error = it.field_key().get(key))) { it.abandon(); return {error, matched}; }
+      if ((error = it.field_value())) { it.abandon(); return {error, matched}; }
+      idx = Selector::match_raw(key);
+    } else {
+      // Otherwise derive the key length from the structural index (the following
+      // ':' token) to feed the length-aware hash, avoiding a forward SIMD scan.
+      std::size_t key_len;
+      if ((error = it.field_key_with_length(key, key_len))) { it.abandon(); return {error, matched}; }
+      if ((error = it.field_value())) { it.abandon(); return {error, matched}; }
+      idx = Selector::match_raw(key.raw(), key_len);
+    }
+    if (idx < Selector::size()) {
+      const std::uint64_t seen_bit = std::uint64_t{1} << (idx & 63);
+      std::uint64_t &seen_word = seen[idx >> 6];
+      if (!(seen_word & seen_bit)) {
+        seen_word |= seen_bit;
+        value matched_value(it.child());
+        // The callback may return void or anything convertible to error_code
+        // (error_code itself, or a for_each_result from a nested for_each). When
+        // it yields an error_code, we stop at the first non-SUCCESS result and
+        // propagate it so the caller can surface value-parse errors (for example,
+        // a type mismatch on a matched field). A void-returning callback is
+        // responsible for handling its own errors.
+        if constexpr (std::is_convertible_v<decltype(on_match(idx, matched_value)), error_code>) {
+          // Unlike the internal-error paths above, a callback error does not
+          // abandon the iterator: we leave it recoverable so the caller can keep
+          // using the object (or its parent) after handling the error.
+          if ((error = on_match(idx, matched_value))) { return {error, matched}; }
+        } else {
+          on_match(idx, matched_value);
+        }
+        if (++matched >= Selector::size()) { break; }
+      }
+    }
+    // Mirror object_iterator::operator++'s safety rail: if the callback consumed
+    // the value and left the iterator closed or in error (e.g. a void callback
+    // that swallowed a fatal sub-iteration error), stop here rather than calling
+    // skip_child on a closed iterator.
+    if (!it.is_open()) { break; }
+    // Skip the value (a no-op if the callback consumed it) and step to the next
+    // field; has_next_field() ends the container on '}', which closes the loop.
+    if ((error = it.skip_child())) { it.abandon(); return {error, matched}; }
+    if ((error = it.has_next_field().error())) { return {error, matched}; }
+  }
+  return {SUCCESS, matched};
+}
+
+namespace key_selector_for_each_detail {
+
+// Dispatch a matched value to the I-th handler in the tuple (0-based). A handler
+// is either an invocable taking the value (void- or error_code-returning) or a
+// deserialization target, in which case we assign via value::get. We always
+// return error_code so the core (index, value) for_each can treat the adapter
+// uniformly.
+template <typename Tuple, std::size_t... Is>
+simdjson_really_inline error_code dispatch_value(
+    std::size_t idx, Tuple& handlers, value v, std::index_sequence<Is...>) {
+  error_code err = SUCCESS;
+  auto try_one = [&](auto Ic) {
+    constexpr std::size_t I = decltype(Ic)::value;
+    if (idx == I) {
+      auto&& h = std::get<I>(handlers);
+      using H = std::remove_reference_t<decltype(h)>;
+      if constexpr (std::is_invocable_v<H&, value>) {
+        // A handler returning void runs for its side effects; one returning
+        // anything convertible to error_code (error_code, or a for_each_result
+        // from a nested for_each) has its error captured and propagated.
+        if constexpr (std::is_convertible_v<decltype(h(v)), error_code>) {
+          err = h(v);
+        } else {
+          h(v);
+        }
+      } else {
+        // Direct deserialization target: assign the matched value into it.
+        err = v.get(h);
+      }
+    }
+  };
+  (try_one(std::integral_constant<std::size_t, Is>{}), ...);
+  return err;
+}
+
+} // namespace key_selector_for_each_detail
+
+template <typename Selector, typename... Handlers>
+  requires key_selector_type<Selector> &&
+           (sizeof...(Handlers) == Selector::size()) &&
+           (key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_flatten simdjson_inline for_each_result object::for_each(Handlers&&... on_match)
+    noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+  auto handlers = std::forward_as_tuple(std::forward<Handlers>(on_match)...);
+  // Reuse the single (index, value) implementation via a tiny adapter.
+  // The adapter is called once per *matched* key (very few); the hot path
+  // (iteration + match_raw + seen bitset) stays exactly the same.
+  return this->template for_each<Selector>(
+      [&](std::size_t i, value v) -> error_code {
+        return key_selector_for_each_detail::dispatch_value(
+            i, handlers, v, std::make_index_sequence<Selector::size()>{});
+      });
+}
+
+template <constevalutil::fixed_string... Keys, typename... Handlers>
+  requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+           (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+           (key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline for_each_result object::for_each(Handlers&&... on_match)
+    noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+  using Selector = key_selector<Keys...>;
+  return this->template for_each<Selector>(std::forward<Handlers>(on_match)...);
+}
+#endif
+
 simdjson_inline simdjson_result<object> object::start(value_iterator &iter) noexcept {
   SIMDJSON_TRY( iter.start_object().error() );
   return object(iter);
@@ -169235,6 +217275,9 @@ simdjson_warn_unused simdjson_inline error_code object::consume() noexcept {
 }

 simdjson_inline simdjson_result<std::string_view> object::raw_json() noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   const uint8_t * starting_point{iter.peek_start()};
   auto error = consume();
   if(error) { return error; }
@@ -169256,9 +217299,17 @@ simdjson_inline object::object(const value_iterator &_iter) noexcept
 {
 }

-simdjson_inline simdjson_result<object_iterator> object::begin() noexcept {
+simdjson_inline simdjson_result<object_iterator> object::begin() & noexcept {
 #if SIMDJSON_DEVELOPMENT_CHECKS
-  if (!iter.is_at_iterator_start()) { return OUT_OF_ORDER_ITERATION; }
+  if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
+  return object_iterator(iter, this);
+#endif
+  return object_iterator(iter);
+}
+simdjson_inline simdjson_result<object_iterator> object::begin() && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  // The object is a temporary that the iterator may outlive: do not lock it.
+  if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
 #endif
   return object_iterator(iter);
 }
@@ -169267,7 +217318,9 @@ simdjson_inline simdjson_result<object_iterator> object::end() noexcept {
 }

 inline simdjson_result<value> object::at_pointer(std::string_view json_pointer) noexcept {
-  if (json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
+  // An empty pointer has no json_pointer[0]: with a default-constructed
+  // std::string_view, reading it dereferences a null pointer.
+  if (json_pointer.empty() || json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
   json_pointer = json_pointer.substr(1);
   size_t slash = json_pointer.find('/');
   std::string_view key = json_pointer.substr(0, slash);
@@ -169369,6 +217422,9 @@ simdjson_inline simdjson_result<size_t> object::count_fields() & noexcept {
 }

 simdjson_inline simdjson_result<bool> object::is_empty() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   bool is_not_empty;
   auto error = iter.reset_object().get(is_not_empty);
   if(error) { return error; }
@@ -169376,9 +217432,41 @@ simdjson_inline simdjson_result<bool> object::is_empty() & noexcept {
 }

 simdjson_inline simdjson_result<bool> object::reset() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   return iter.reset_object();
 }

+simdjson_inline object_position object::get_current_position() const noexcept {
+  return object_position{iter.position(), iter.json_iter().depth()};
+}
+
+simdjson_inline error_code object::revert_position(object_position position) noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
+  // json_iterator::reenter_child() requires the live depth to be exactly
+  // one level shallower than the target (matching how every other depth
+  // transition in this iterator works), and under SIMDJSON_DEVELOPMENT_CHECKS
+  // additionally validates against the parser's per-depth container-start
+  // bookkeeping. Neither applies here: depending on what was captured and
+  // what has happened since (a scalar field fully consumed, a compound
+  // value left open, a find_field() miss that scanned past everything),
+  // the live depth when reverting can be any number of levels away from
+  // the captured one, and the captured depth is not necessarily a
+  // container's own start. reenter_at() moves directly, matching how
+  // reset_object() itself repositions without going through reenter_child().
+  iter.reenter_at(position.position, position.depth);
+  return SUCCESS;
+}
+
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline void object::set_locked(bool _locked) noexcept {
+  locked = _locked;
+}
+#endif
+
 #if SIMDJSON_SUPPORTS_CONCEPTS && SIMDJSON_STATIC_REFLECTION

 template<constevalutil::fixed_string... FieldNames, typename T>
@@ -169436,10 +217524,14 @@ simdjson_inline simdjson_result<lasx::ondemand::object>::simdjson_result(lasx::o
 simdjson_inline simdjson_result<lasx::ondemand::object>::simdjson_result(error_code error) noexcept
     : implementation_simdjson_result_base<lasx::ondemand::object>(error) {}

-simdjson_inline simdjson_result<lasx::ondemand::object_iterator> simdjson_result<lasx::ondemand::object>::begin() noexcept {
+simdjson_inline simdjson_result<lasx::ondemand::object_iterator> simdjson_result<lasx::ondemand::object>::begin() & noexcept {
   if (error()) { return error(); }
   return first.begin();
 }
+simdjson_inline simdjson_result<lasx::ondemand::object_iterator> simdjson_result<lasx::ondemand::object>::begin() && noexcept {
+  if (error()) { return error(); }
+  return std::move(first).begin();
+}
 simdjson_inline simdjson_result<lasx::ondemand::object_iterator> simdjson_result<lasx::ondemand::object>::end() noexcept {
   if (error()) { return error(); }
   return first.end();
@@ -169493,11 +217585,55 @@ simdjson_inline error_code simdjson_result<lasx::ondemand::object>::for_each_at_
   return first.for_each_at_path_with_wildcard(json_path, std::forward<Func>(callback));
 }

+#if SIMDJSON_SUPPORTS_CONCEPTS
+template <typename Selector, typename Func>
+  requires lasx::ondemand::key_selector_type<Selector> &&
+           std::is_invocable_v<Func&, std::size_t, lasx::ondemand::value>
+simdjson_inline lasx::ondemand::for_each_result
+simdjson_result<lasx::ondemand::object>::for_each(Func&& on_match)
+    noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, lasx::ondemand::value>) {
+  if (error()) { return {error(), 0}; }
+  return first.template for_each<Selector>(std::forward<Func>(on_match));
+}
+
+template <typename Selector, typename... Handlers>
+  requires lasx::ondemand::key_selector_type<Selector> &&
+           (sizeof...(Handlers) == Selector::size()) &&
+           (lasx::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline lasx::ondemand::for_each_result
+simdjson_result<lasx::ondemand::object>::for_each(Handlers&&... on_match)
+    noexcept(lasx::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+  if (error()) { return {error(), 0}; }
+  return first.template for_each<Selector>(std::forward<Handlers>(on_match)...);
+}
+
+template <constevalutil::fixed_string... Keys, typename... Handlers>
+  requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+           (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+           (lasx::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline lasx::ondemand::for_each_result
+simdjson_result<lasx::ondemand::object>::for_each(Handlers&&... on_match)
+    noexcept(lasx::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+  if (error()) { return {error(), 0}; }
+  return first.template for_each<Keys...>(std::forward<Handlers>(on_match)...);
+}
+#endif
+
 inline simdjson_result<bool> simdjson_result<lasx::ondemand::object>::reset() noexcept {
   if (error()) { return error(); }
   return first.reset();
 }

+inline simdjson_result<lasx::ondemand::object_position> simdjson_result<lasx::ondemand::object>::get_current_position() noexcept {
+  if (error()) { return error(); }
+  return first.get_current_position();
+}
+
+inline error_code simdjson_result<lasx::ondemand::object>::revert_position(lasx::ondemand::object_position position) noexcept {
+  if (error()) { return error(); }
+  return first.revert_position(position);
+}
+
 inline simdjson_result<bool> simdjson_result<lasx::ondemand::object>::is_empty() noexcept {
   if (error()) { return error(); }
   return first.is_empty();
@@ -169541,6 +217677,61 @@ simdjson_inline object_iterator::object_iterator(const value_iterator &_iter) no
   : iter{_iter}
 {}

+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline object_iterator::object_iterator(const value_iterator &_iter, object* _parent) noexcept
+  : parent{_parent}, iter{_iter}
+{
+  if (parent) parent->set_locked(true);
+}
+#endif
+
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline object_iterator::~object_iterator() noexcept
+{
+  if (parent) parent->set_locked(false);
+}
+
+simdjson_inline object_iterator::object_iterator(object_iterator&& other) noexcept
+  : has_been_referenced{other.has_been_referenced},
+    parent{other.parent},
+    iter{std::move(other.iter)}
+{
+  other.parent = nullptr;
+}
+
+simdjson_inline object_iterator& object_iterator::operator=(object_iterator&& other) noexcept {
+  if (this != &other)
+  {
+    if (parent)
+      parent->set_locked(false);
+    has_been_referenced = other.has_been_referenced;
+    parent = other.parent;
+    iter = std::move(other.iter);
+
+    other.parent = nullptr;
+  }
+  return *this;
+}
+
+simdjson_inline object_iterator::object_iterator(const object_iterator& other) noexcept
+  : has_been_referenced{other.has_been_referenced},
+    parent{nullptr},
+    iter{other.iter}
+{}
+
+simdjson_inline object_iterator& object_iterator::operator=(const object_iterator& other) noexcept {
+  if (this != &other)
+  {
+    if (parent)
+      parent->set_locked(false);
+    has_been_referenced = other.has_been_referenced;
+    parent = nullptr;
+    iter = other.iter;
+  }
+  return *this;
+}
+#endif
+
 simdjson_inline simdjson_result<field> object_iterator::operator*() noexcept {
 #if SIMDJSON_DEVELOPMENT_CHECKS
   // We must call * once per iteration.
@@ -169668,6 +217859,147 @@ simdjson_inline simdjson_result<lasx::ondemand::object_iterator> &simdjson_resul

 #endif // SIMDJSON_GENERIC_ONDEMAND_OBJECT_ITERATOR_INL_H
 /* end file simdjson/generic/ondemand/object_iterator-inl.h for lasx */
+/* including simdjson/generic/ondemand/ranges-inl.h for lasx: #include "simdjson/generic/ondemand/ranges-inl.h" */
+/* begin file simdjson/generic/ondemand/ranges-inl.h for lasx */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/ranges.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array_iterator-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object_iterator-inl.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#if SIMDJSON_SUPPORTS_RANGES
+
+namespace simdjson {
+namespace lasx {
+namespace ondemand {
+
+//
+// array_range_iterator
+//
+
+simdjson_inline array_range_iterator::array_range_iterator(array_iterator iter) noexcept
+  : iter_{iter} {}
+
+simdjson_inline simdjson_result<value> array_range_iterator::operator*() const noexcept {
+  return *iter_;
+}
+
+simdjson_inline array_range_iterator& array_range_iterator::operator++() noexcept {
+  ++iter_;
+  return *this;
+}
+
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+simdjson_inline void array_range_iterator::operator++(int) noexcept {
+  ++*this;
+}
+SIMDJSON_POP_DISABLE_WARNINGS
+
+//
+// array_range
+//
+
+simdjson_inline array_range::array_range(array& arr) noexcept {
+  auto b = arr.begin();
+  if (b.error()) { error_ = b.error(); return; }
+  begin_ = b.value_unsafe();
+  end_ = arr.end().value_unsafe();
+}
+
+simdjson_inline array_range_iterator array_range::begin() noexcept {
+  return array_range_iterator(begin_);
+}
+
+simdjson_inline array_range_iterator array_range::end() noexcept {
+  return array_range_iterator(end_);
+}
+
+//
+// object_range_iterator
+//
+
+simdjson_inline object_range_iterator::object_range_iterator(object_iterator iter) noexcept
+  : iter_{iter} {}
+
+simdjson_inline simdjson_result<field> object_range_iterator::operator*() const noexcept {
+  return *iter_;
+}
+
+simdjson_inline object_range_iterator& object_range_iterator::operator++() noexcept {
+  ++iter_;
+  return *this;
+}
+
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+simdjson_inline void object_range_iterator::operator++(int) noexcept {
+  ++*this;
+}
+SIMDJSON_POP_DISABLE_WARNINGS
+
+
+//
+// object_range
+//
+
+simdjson_inline object_range::object_range(object& obj) noexcept {
+  auto b = obj.begin();
+  if (b.error()) { error_ = b.error(); return; }
+  begin_ = b.value_unsafe();
+  end_ = obj.end().value_unsafe();
+}
+
+simdjson_inline object_range_iterator object_range::begin() noexcept {
+  return object_range_iterator(begin_);
+}
+
+simdjson_inline object_range_iterator object_range::end() noexcept {
+  return object_range_iterator(end_);
+}
+
+//
+// Free functions
+//
+
+simdjson_inline array_range get_range(array& arr) noexcept {
+  return array_range(arr);
+}
+
+simdjson_inline object_range get_key_value_range(object& obj) noexcept {
+  return object_range(obj);
+}
+
+#if SIMDJSON_EXCEPTIONS
+simdjson_inline array_range get_range(simdjson_result<array> result) {
+  return array_range(result.value());
+}
+
+simdjson_inline object_range get_key_value_range(simdjson_result<object> result) {
+  return object_range(result.value());
+}
+#endif // SIMDJSON_EXCEPTIONS
+
+} // namespace ondemand
+} // namespace lasx
+} // namespace simdjson
+
+// Verify the range wrapper types satisfy the expected C++20 concepts.
+static_assert(std::input_iterator<simdjson::lasx::ondemand::array_range_iterator>);
+static_assert(std::input_iterator<simdjson::lasx::ondemand::object_range_iterator>);
+static_assert(std::ranges::input_range<simdjson::lasx::ondemand::array_range>);
+static_assert(std::ranges::input_range<simdjson::lasx::ondemand::object_range>);
+static_assert(std::ranges::view<simdjson::lasx::ondemand::array_range>);
+static_assert(std::ranges::view<simdjson::lasx::ondemand::object_range>);
+
+#endif // SIMDJSON_SUPPORTS_RANGES
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H
+/* end file simdjson/generic/ondemand/ranges-inl.h for lasx */
 /* including simdjson/generic/ondemand/parser-inl.h for lasx: #include "simdjson/generic/ondemand/parser-inl.h" */
 /* begin file simdjson/generic/ondemand/parser-inl.h for lasx */
 #ifndef SIMDJSON_GENERIC_ONDEMAND_PARSER_INL_H
@@ -169699,7 +218031,10 @@ simdjson_warn_unused simdjson_inline error_code parser::allocate(size_t new_capa

   // string_capacity copied from document::allocate
   _capacity = 0;
-  size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * new_capacity / 3 + SIMDJSON_PADDING, 64);
+  if(5 * (new_capacity / 3) + SIMDJSON_PADDING < SIMDJSON_PADDING) {
+    return CAPACITY; // overflow, only happen on legacy 32-bit systems with very large capacity
+  }
+  size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * (new_capacity / 3) + SIMDJSON_PADDING, 64);
   string_buf.reset(new (std::nothrow) uint8_t[string_capacity]);
 #if SIMDJSON_DEVELOPMENT_CHECKS
   start_positions.reset(new (std::nothrow) token_position[new_max_depth]);
@@ -169724,6 +218059,7 @@ simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate(p
   if (!json.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }

   json.remove_utf8_bom();
+  _document_len = json.length();

   // Allocate if needed
   if (capacity() < json.length() || !string_buf) {
@@ -169740,6 +218076,7 @@ simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate_a
   if (!json.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }

   json.remove_utf8_bom();
+  _document_len = json.length();

   // Allocate if needed
   if (capacity() < json.length() || !string_buf) {
@@ -169805,6 +218142,34 @@ simdjson_warn_unused simdjson_inline simdjson_result<json_iterator> parser::iter
   return json_iterator(reinterpret_cast<const uint8_t *>(json.data()), this);
 }

+inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size) noexcept {
+  if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
+  if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+    buf += 3;
+    len -= 3;
+  }
+  return document_stream(*this, buf, len, batch_size, false, stream_format::whitespace_delimited);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size) noexcept {
+  return iterate_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(padded_string_view s, size_t batch_size) noexcept {
+  if (!s.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
+  return iterate_many(s.data(), s.length(), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const padded_string &s, size_t batch_size) noexcept {
+  return iterate_many(padded_string_view(s), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const std::string &s, size_t batch_size) noexcept {
+  return iterate_many(padded_string_view(s), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size) noexcept {
+  return iterate_many(pad(s), batch_size);
+}
+
+#ifndef SIMDJSON_DISABLE_DEPRECATED_API
+SIMDJSON_PUSH_DISABLE_WARNINGS
+SIMDJSON_DISABLE_DEPRECATED_WARNING
 inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept {
   // Warning: no check is done on the buffer padding. We trust the user.
   if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
@@ -169812,8 +218177,11 @@ inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf,
     buf += 3;
     len -= 3;
   }
-  if(allow_comma_separated && batch_size < len) { batch_size = len; }
-  return document_stream(*this, buf, len, batch_size, allow_comma_separated);
+  // Map allow_comma_separated to stream_format::comma_delimited
+  if (allow_comma_separated) {
+    return document_stream(*this, buf, len, batch_size, false, stream_format::comma_delimited);
+  }
+  return document_stream(*this, buf, len, batch_size, false, stream_format::whitespace_delimited);
 }

 inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept {
@@ -169833,6 +218201,51 @@ inline simdjson_result<document_stream> parser::iterate_many(const std::string &
 inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size, bool allow_comma_separated) noexcept {
   return iterate_many(pad(s), batch_size, allow_comma_separated);
 }
+SIMDJSON_POP_DISABLE_WARNINGS
+#endif // SIMDJSON_DISABLE_DEPRECATED_API
+
+inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+  if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
+  if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+    buf += 3;
+    len -= 3;
+  }
+  if (format == stream_format::comma_delimited_array) {
+    // Strip leading JSON whitespace.
+    while (len > 0 && (buf[0] == ' ' || buf[0] == '\t' || buf[0] == '\n' || buf[0] == '\r')) {
+      buf++; len--;
+    }
+    // Expect the opening '['.
+    if (len == 0 || buf[0] != '[') { return TAPE_ERROR; }
+    buf++; len--;
+    // Strip trailing JSON whitespace.
+    while (len > 0 && (buf[len-1] == ' ' || buf[len-1] == '\t' || buf[len-1] == '\n' || buf[len-1] == '\r')) {
+      len--;
+    }
+    // Expect the closing ']'.
+    if (len == 0 || buf[len-1] != ']') { return TAPE_ERROR; }
+    len--;
+    // Fall through to comma_delimited over the array contents.
+    format = stream_format::comma_delimited;
+  }
+  return document_stream(*this, buf, len, batch_size, false, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+  return iterate_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(padded_string_view s, size_t batch_size, stream_format format) noexcept {
+  if (!s.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
+  return iterate_many(s.data(), s.length(), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const std::string &s, size_t batch_size, stream_format format) noexcept {
+  return iterate_many(padded_string_view(s), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size, stream_format format) noexcept {
+  return iterate_many(pad(s), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const padded_string &s, size_t batch_size, stream_format format) noexcept {
+  return iterate_many(padded_string_view(s), batch_size, format);
+}
 simdjson_pure simdjson_inline size_t parser::capacity() const noexcept {
   return _capacity;
 }
@@ -170240,6 +218653,27 @@ namespace simdjson {
 namespace lasx {
 namespace ondemand {

+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+simdjson_inline bool raw_json_string_is_quote_terminated(const uint8_t *json, uint32_t max_len) noexcept {
+  bool escaping{false};
+  for (uint32_t i = 1; i < max_len; i++) {
+    switch (json[i]) {
+      case '"':
+        if (!escaping) { return true; }
+        escaping = false;
+        break;
+      case '\\':
+        escaping = !escaping;
+        break;
+      default:
+        escaping = false;
+        break;
+    }
+  }
+  return false;
+}
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+
 simdjson_inline value_iterator::value_iterator(
   json_iterator *json_iter,
   depth_t depth,
@@ -170627,6 +219061,22 @@ simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iter
   return raw_json_string(key);
 }

+simdjson_warn_unused simdjson_inline error_code value_iterator::field_key_with_length(raw_json_string &key, std::size_t &len) noexcept {
+  assert_at_next();
+
+  const uint8_t *k = _json_iter->return_current_and_advance();
+  if (*(k++) != '"') { return report_error(TAPE_ERROR, "Object key is not a string"); }
+  // After return_current_and_advance(), the current token is the ':' that follows
+  // the key. The closing quote sits just before it (only JSON whitespace may
+  // intervene), so step back from the ':' to the closing quote to get the length.
+  // In minified JSON this is a single back-step.
+  const char *q = reinterpret_cast<const char *>(_json_iter->peek());
+  do { --q; } while (*q != '"');
+  key = raw_json_string(k);
+  len = static_cast<std::size_t>(q - reinterpret_cast<const char *>(k));
+  return SUCCESS;
+}
+
 simdjson_warn_unused simdjson_inline error_code value_iterator::field_value() noexcept {
   assert_at_next();

@@ -170744,7 +219194,7 @@ simdjson_warn_unused simdjson_inline error_code value_iterator::get_string(strin
   auto saved_string_buf_loc = _json_iter->string_buf_loc();
   auto err = get_string(allow_replacement).get(content);
   if (err) { return err; }
-  receiver = content;
+  internal::assign_utf8(receiver, content);
   // Restore the string buffer location, effectively discarding any temporary string storage
   _json_iter->string_buf_loc() = saved_string_buf_loc;
   return SUCCESS;
@@ -170755,6 +219205,15 @@ simdjson_warn_unused simdjson_inline simdjson_result<std::string_view> value_ite
 simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iterator::get_raw_json_string() noexcept {
   auto json = peek_scalar("string");
   if (*json != '"') { return incorrect_type_error("Not a string"); }
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  if (_json_iter->allow_incomplete_json()) {
+    const size_t remaining_input_length = _json_iter->remaining_input_length(json);
+    const uint32_t max_len = remaining_input_length < peek_start_length() ? uint32_t(remaining_input_length) : peek_start_length();
+    if (!raw_json_string_is_quote_terminated(json, max_len)) {
+      return STRING_ERROR;
+    }
+  }
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
   advance_scalar("string");
   return raw_json_string(json+1);
 }
@@ -170788,6 +219247,16 @@ simdjson_warn_unused simdjson_inline simdjson_result<double> value_iterator::get
   if(result.error() == SUCCESS) { advance_non_root_scalar("double"); }
   return result;
 }
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_float() noexcept {
+  auto result = numberparsing::parse_float(peek_non_root_scalar("float"));
+  if(result.error() == SUCCESS) { advance_non_root_scalar("float"); }
+  return result;
+}
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_float_in_string() noexcept {
+  auto result = numberparsing::parse_float_in_string(peek_non_root_scalar("float"));
+  if(result.error() == SUCCESS) { advance_non_root_scalar("float"); }
+  return result;
+}
 simdjson_warn_unused simdjson_inline simdjson_result<bool> value_iterator::get_bool() noexcept {
   auto result = parse_bool(peek_non_root_scalar("bool"));
   if(result.error() == SUCCESS) { advance_non_root_scalar("bool"); }
@@ -170890,7 +219359,7 @@ simdjson_warn_unused simdjson_inline error_code value_iterator::get_root_string(
   auto saved_string_buf_loc = _json_iter->string_buf_loc();
   auto err = get_root_string(check_trailing, allow_replacement).get(content);
   if (err) { return err; }
-  receiver = content;
+  internal::assign_utf8(receiver, content);
   // Restore the string buffer location, effectively discarding any temporary string storage
   _json_iter->string_buf_loc() = saved_string_buf_loc;
   return SUCCESS;
@@ -170902,6 +219371,15 @@ simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iter
   auto json = peek_scalar("string");
   if (*json != '"') { return incorrect_type_error("Not a string"); }
   if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  if (_json_iter->allow_incomplete_json()) {
+    const size_t remaining_input_length = _json_iter->remaining_input_length(json);
+    const uint32_t max_len = remaining_input_length < peek_root_length() ? uint32_t(remaining_input_length) : peek_root_length();
+    if (!raw_json_string_is_quote_terminated(json, max_len)) {
+      return STRING_ERROR;
+    }
+  }
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
   advance_scalar("string");
   return raw_json_string(json+1);
 }
@@ -171011,6 +219489,43 @@ simdjson_warn_unused simdjson_inline simdjson_result<double> value_iterator::get
   return result;
 }

+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_root_float(bool check_trailing) noexcept {
+  auto max_len = peek_root_length();
+  auto json = peek_root_scalar("float");
+  // We use the same buffer size as get_root_double: the number of significant
+  // digits that matter is smaller for binary32, but the JSON document may still
+  // spell out a long number that we must parse (and round) faithfully.
+  uint8_t tmpbuf[1074+8+1+1]; // +1 for null termination.
+  tmpbuf[1074+8+1] = '\0'; // make sure that buffer is always null terminated.
+  if (!_json_iter->copy_to_buffer(json, max_len, tmpbuf, 1074+8+1)) {
+    logger::log_error(*_json_iter, start_position(), depth(), "Root number more than 1082 characters");
+    return NUMBER_ERROR;
+  }
+  auto result = numberparsing::parse_float(tmpbuf);
+  if(result.error() == SUCCESS) {
+    if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+    advance_root_scalar("float");
+  }
+  return result;
+}
+
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_root_float_in_string(bool check_trailing) noexcept {
+  auto max_len = peek_root_length();
+  auto json = peek_root_scalar("float");
+  uint8_t tmpbuf[1074+8+1+1]; // +1 for null termination.
+  tmpbuf[1074+8+1] = '\0'; // make sure that buffer is always null terminated.
+  if (!_json_iter->copy_to_buffer(json, max_len, tmpbuf, 1074+8+1)) {
+    logger::log_error(*_json_iter, start_position(), depth(), "Root number more than 1082 characters");
+    return NUMBER_ERROR;
+  }
+  auto result = numberparsing::parse_float_in_string(tmpbuf);
+  if(result.error() == SUCCESS) {
+    if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+    advance_root_scalar("float");
+  }
+  return result;
+}
+
 simdjson_warn_unused simdjson_inline simdjson_result<bool> value_iterator::get_root_bool(bool check_trailing) noexcept {
   auto max_len = peek_root_length();
   auto json = peek_root_scalar("bool");
@@ -171239,6 +219754,21 @@ simdjson_inline void value_iterator::move_at_container_start() noexcept {
   _json_iter->token.set_position(_start_position + 1);
 }

+simdjson_inline void value_iterator::reenter_at(token_position position, depth_t depth) noexcept {
+  // Unlike reenter_child(), this does not require the live depth to be
+  // exactly one level shallower than depth, nor does it validate against
+  // the parser's per-depth container-start bookkeeping: neither holds in
+  // general for a caller-supplied snapshot (see object_position). What
+  // must still always hold, regardless of what was captured or how far
+  // the live iterator has since moved, is that position and depth are
+  // themselves sane values -- this is the same bound reenter_child()
+  // itself applies unconditionally.
+  SIMDJSON_ASSUME(position != nullptr);
+  SIMDJSON_ASSUME(depth >= 1 && depth < INT32_MAX);
+  _json_iter->_depth = depth;
+  _json_iter->token.set_position(position);
+}
+
 simdjson_inline simdjson_result<bool> value_iterator::reset_array() noexcept {
   if(error()) { return error(); }
   move_at_container_start();
@@ -172641,11 +221171,6 @@ simdjson_inline long long int count_ones(uint64_t input_num) {
   return __builtin_popcountll(input_num);
 }

-simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
-                                uint64_t *result) {
-  return __builtin_uaddll_overflow(value1, value2,
-                                   reinterpret_cast<unsigned long long *>(result));
-}

 } // unnamed namespace
 } // namespace rvv_vls
@@ -173193,7 +221718,7 @@ simdjson_inline internal::value128 full_multiplication(uint64_t value1, uint64_t
 /* end file simdjson/rvv-vls/begin.h */
 /* including simdjson/generic/ondemand/amalgamated.h for rvv_vls: #include "simdjson/generic/ondemand/amalgamated.h" */
 /* begin file simdjson/generic/ondemand/amalgamated.h for rvv_vls */
-#if defined(SIMDJSON_CONDITIONAL_INCLUDE) && !defined(SIMDJSON_GENERIC_BUILDER_DEPENDENCIES_H)
+#if defined(SIMDJSON_CONDITIONAL_INCLUDE) && !defined(SIMDJSON_GENERIC_ONDEMAND_DEPENDENCIES_H)
 #error simdjson/generic/ondemand/dependencies.h must be included before simdjson/generic/ondemand/amalgamated.h!
 #endif

@@ -173242,6 +221767,13 @@ class token_iterator;
 class value;
 class value_iterator;

+#if SIMDJSON_SUPPORTS_RANGES
+class array_range;
+class array_range_iterator;
+class object_range;
+class object_range_iterator;
+#endif // SIMDJSON_SUPPORTS_RANGES
+
 } // namespace ondemand
 } // namespace rvv_vls
 } // namespace simdjson
@@ -173274,6 +221806,9 @@ template <> struct is_builtin_deserializable<rvv_vls::ondemand::object> : std::t
 template <> struct is_builtin_deserializable<rvv_vls::ondemand::value> : std::true_type {};
 template <> struct is_builtin_deserializable<rvv_vls::ondemand::raw_json_string> : std::true_type {};
 template <> struct is_builtin_deserializable<std::string_view> : std::true_type {};
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template <> struct is_builtin_deserializable<std::u8string_view> : std::true_type {};
+#endif // SIMDJSON_SUPPORTS_CHAR8_T

 template <typename T>
 concept is_builtin_deserializable_v = is_builtin_deserializable<T>::value;
@@ -173291,6 +221826,10 @@ concept nothrow_custom_deserializable = nothrow_tag_invocable<deserialize_tag, V
 template <typename T, typename ValT = rvv_vls::ondemand::value>
 concept nothrow_deserializable = nothrow_custom_deserializable<T, ValT> || is_builtin_deserializable_v<T>;

+// True iff get<T>() on ValT cannot throw: no custom tag_invoke, or that tag_invoke is noexcept.
+template <typename T, typename ValT = rvv_vls::ondemand::value>
+concept nothrow_gettable = !custom_deserializable<T, ValT> || nothrow_custom_deserializable<T, ValT>;
+
 /// Deserialize Tag
 inline constexpr struct deserialize_tag {
   using array_type = rvv_vls::ondemand::array;
@@ -173505,6 +222044,17 @@ public:
    */
   simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> field_key() noexcept;

+  /**
+   * Get the current field's key together with its raw byte length.
+   *
+   * Like field_key(), but also returns the number of raw key bytes (the distance
+   * from the first key byte to the closing quote). The length is recovered from
+   * the structural index -- the next structural token is the ':' -- by stepping
+   * back over any whitespace to the closing quote, avoiding a forward SIMD scan
+   * for the closing quote. Leaves the iterator positioned exactly as field_key().
+   */
+  simdjson_warn_unused simdjson_inline error_code field_key_with_length(raw_json_string &key, std::size_t &len) noexcept;
+
   /**
    * Pass the : in the field and move to its value.
    */
@@ -173657,6 +222207,8 @@ public:
   simdjson_warn_unused simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_warn_unused simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_warn_unused simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<bool> get_bool() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<bool> is_null() noexcept;
   simdjson_warn_unused simdjson_inline bool is_negative() noexcept;
@@ -173675,6 +222227,8 @@ public:
   simdjson_warn_unused simdjson_inline simdjson_result<int64_t> get_root_int64_in_string(bool check_trailing) noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<double> get_root_double(bool check_trailing) noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<double> get_root_double_in_string(bool check_trailing) noexcept;
+  simdjson_warn_unused simdjson_inline simdjson_result<float> get_root_float(bool check_trailing) noexcept;
+  simdjson_warn_unused simdjson_inline simdjson_result<float> get_root_float_in_string(bool check_trailing) noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<bool> get_root_bool(bool check_trailing) noexcept;
   simdjson_warn_unused simdjson_inline bool is_root_negative() noexcept;
   simdjson_warn_unused simdjson_inline simdjson_result<bool> is_root_integer(bool check_trailing) noexcept;
@@ -173810,6 +222364,15 @@ protected:

   /** @copydoc error_code json_iterator::position() const noexcept; */
   simdjson_inline token_position position() const noexcept;
+  /**
+   * Move the live iterator directly to the given position and depth, without
+   * validating against the parser's per-depth container-start bookkeeping
+   * (unlike json_iterator::reenter_child()). Used to restore a previously
+   * captured mid-container position (see object::revert_position()): that
+   * bookkeeping only tracks each container's own start, not every position
+   * a caller might later capture and revert to, so it does not apply here.
+   */
+  simdjson_inline void reenter_at(token_position position, depth_t depth) noexcept;
   /** @copydoc error_code json_iterator::end_position() const noexcept; */
   simdjson_inline token_position last_position() const noexcept;
   /** @copydoc error_code json_iterator::end_position() const noexcept; */
@@ -173878,9 +222441,12 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+   * Supported types: object, array, raw_json_string, string_view, u8string_view (C++20), uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool
    *
-   * You may use get_double(), get_bool(), get_uint64(), get_int64(),
+   * You may use get_double(), get_float(), get_bool(), get_uint64(), get_int64(),
+   * get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(),
+   * get_float32(), get_float64(),
    * get_object(), get_array(), get_raw_json_string(), or get_string() instead.
    * When SIMDJSON_SUPPORTS_CONCEPTS is set, custom types are also supported.
    *
@@ -173890,7 +222456,7 @@ public:
   template <typename T>
   simdjson_inline simdjson_result<T> get()
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+    noexcept(nothrow_gettable<T, value>)
 #else
     noexcept
 #endif
@@ -173905,7 +222471,8 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+   * Supported types: object, array, raw_json_string, string_view, u8string_view (C++20), uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool
    * If the macro SIMDJSON_SUPPORTS_CONCEPTS is set, then custom types are also supported.
    *
    * @param out This is set to a value of the given type, parsed from the JSON. If there is an error, this may not be initialized.
@@ -173915,7 +222482,7 @@ public:
   template <typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out)
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+    noexcept(nothrow_gettable<T, value>)
 #else
     noexcept
 #endif
@@ -173943,7 +222510,7 @@ public:
       "And you do not seem to have added support for it. Indeed, we have that "
       "simdjson::custom_deserializable<T> is false and the type T is not a default type "
       "such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-      "int64_t, double, or bool.");
+      "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
     static_cast<void>(out); // to get rid of unused errors
     return UNINITIALIZED;
   }
@@ -173952,7 +222519,8 @@ public:
     // immediately fail.
     static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
       "The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-      "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+      "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+      "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
       " get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
       " You may also add support for custom types, see our documentation.");
     static_cast<void>(out); // to get rid of unused errors
@@ -174030,6 +222598,50 @@ public:
    */
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;

+  /**
+   * Cast this JSON value to a 16-bit unsigned integer.
+   *
+   * Calls get_uint64() and checks that the result fits in a uint16_t.
+   *
+   * @returns A 16-bit unsigned integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint16_t.
+   */
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+
+  /**
+   * Cast this JSON value to a 16-bit signed integer.
+   *
+   * Calls get_int64() and checks that the result fits in an int16_t.
+   *
+   * @returns A 16-bit signed integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int16_t.
+   */
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+
+  /**
+   * Cast this JSON value to an 8-bit unsigned integer.
+   *
+   * Calls get_uint64() and checks that the result fits in a uint8_t.
+   *
+   * @returns An 8-bit unsigned integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint8_t.
+   */
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+
+  /**
+   * Cast this JSON value to an 8-bit signed integer.
+   *
+   * Calls get_int64() and checks that the result fits in an int8_t.
+   *
+   * @returns An 8-bit signed integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int8_t.
+   */
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
+
   /**
    * Cast this JSON value to a double.
    *
@@ -174046,6 +222658,53 @@ public:
    */
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;

+  /**
+   * Cast this JSON value to a float (binary32).
+   *
+   * The value is rounded directly to binary32: it is the float nearest to the
+   * JSON number. Note that this may differ from get_double() followed by a cast
+   * to float, which rounds twice.
+   *
+   * @returns A float.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+
+  /**
+   * Cast this JSON value (inside string) to a float (binary32).
+   *
+   * @returns A float.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  /**
+   * Cast this JSON value to a std::float32_t (C++23).
+   *
+   * Same as get_float(): the value is rounded directly to binary32.
+   *
+   * @returns A std::float32_t.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  /**
+   * Cast this JSON value to a std::float64_t (C++23).
+   *
+   * Same as get_double().
+   *
+   * @returns A std::float64_t.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   */
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
+
   /**
    * Cast this JSON value to a string.
    *
@@ -174073,6 +222732,26 @@ public:
    */
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;

+#if SIMDJSON_SUPPORTS_CHAR8_T
+  /**
+   * Cast this JSON value to a C++20 UTF-8 string.
+   *
+   * The string is guaranteed to be valid UTF-8.
+   *
+   * Equivalent to get<std::u8string_view>(). This is a zero-copy alias of
+   * get_string(): the very same bytes are returned, viewed as char8_t.
+   *
+   * Important: a value should be consumed once. Calling get_u8string() twice on the same
+   * value is an error.
+   *
+   * @param allow_replacement Whether to allow a replacement character for unmatched surrogate pairs.
+   * @returns An UTF-8 string. The string is stored in the parser and will be invalidated the next
+   *          time it parses a document or when it is destroyed.
+   * @returns INCORRECT_TYPE if the JSON value is not a string.
+   */
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
   /**
    * Attempts to fill the provided std::string reference with the parsed value of the current string.
    *
@@ -174160,7 +222839,7 @@ public:
   /**
    * Cast this JSON value to an unsigned integer.
    *
-   * @returns A signed 64-bit integer.
+   * @returns A unsigned 64-bit integer.
    * @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit unsigned integer.
    */
   simdjson_inline operator uint64_t() noexcept(false);
@@ -174625,9 +223304,24 @@ public:
   simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template <typename string_type>
   simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -174635,9 +223329,19 @@ public:
   simdjson_inline simdjson_result<bool> get_bool() noexcept;
   simdjson_inline simdjson_result<bool> is_null() noexcept;

-  template<typename T> simdjson_inline simdjson_result<T> get() noexcept;
+  template<typename T> simdjson_inline simdjson_result<T> get()
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, rvv_vls::ondemand::value>);
+#else
+    noexcept;
+#endif

-  template<typename T> simdjson_inline error_code get(T &out) noexcept;
+  template<typename T> simdjson_inline error_code get(T &out)
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, rvv_vls::ondemand::value>);
+#else
+    noexcept;
+#endif

 #if SIMDJSON_EXCEPTIONS
   template <class T>
@@ -174968,6 +223672,7 @@ protected:
   token_position _position{};

   friend class json_iterator;
+  friend class document_stream;
   friend class value_iterator;
   friend class object;
   template <typename... Args>
@@ -175059,6 +223764,9 @@ protected:
    * value of this attribute.
    */
   bool _streaming{false};
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  bool _allow_incomplete_json{false};
+#endif

 public:
   simdjson_inline json_iterator() noexcept = default;
@@ -175083,6 +223791,10 @@ public:
    * start_root_array() and start_root_object().
    */
   simdjson_inline bool streaming() const noexcept;
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  simdjson_inline bool allow_incomplete_json() const noexcept;
+  simdjson_inline size_t remaining_input_length(const uint8_t *json) const noexcept;
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON

   /**
    * Get the root value iterator
@@ -175962,33 +224674,87 @@ public:
    * @param batch_size The batch size to use. MUST be larger than the largest document. The sweet
    *                   spot is cache-related: small enough to fit in cache, yet big enough to
    *                   parse as many documents as possible in one tight loop.
-   *                   Defaults to 10MB, which has been a reasonable sweet spot in our tests.
-   * @param allow_comma_separated (defaults on false) This allows a mode where the documents are
-   *                   separated by commas instead of whitespace. It comes with a performance
-   *                   penalty because the entire document is indexed at once (and the document must be
-   *                   less than 4 GB), and there is no multithreading. In this mode, the batch_size parameter
-   *                   is effectively ignored, as it is set to at least the document size.
+   *                   Defaults to 1MB, which has been a reasonable sweet spot in our tests.
+   * @param allow_comma_separated @deprecated Use stream_format::comma_delimited instead.
+   *                   When true, maps internally to stream_format::comma_delimited.
+   *                   Defaults to false.
    * @return The stream, or an error. An empty input will yield 0 documents rather than an EMPTY error. Errors:
    *         - MEMALLOC if the parser does not have enough capacity and memory allocation fails
    *         - CAPACITY if the parser does not have enough capacity and batch_size > max_capacity.
    *         - other json errors if parsing fails. You should not rely on these errors to always the same for the
    *           same document: they may vary under runtime dispatch (so they may vary depending on your system and hardware).
    */
-  inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
-  inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
-  inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
-  inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size)
+  inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size)
     the string might be automatically padded with up to SIMDJSON_PADDING whitespace characters */
-  inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
-  /** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
-  inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size = DEFAULT_BATCH_SIZE, bool allow_comma_separated = false) noexcept;
+  inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept;
+  /** @private An rvalue input is destroyed at the end of the full-expression, while the
+   * returned document_stream only holds a pointer to it: iterating the stream would then
+   * read freed memory. These deleted overloads also catch a std::string_view argument,
+   * which would otherwise convert implicitly to a padded_string temporary. */
+  inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size = DEFAULT_BATCH_SIZE) = delete;// unsafe
+  /** @private @overload iterate_many(const std::string &&s, size_t batch_size) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size = DEFAULT_BATCH_SIZE) = delete;// unsafe
   /** @private We do not want to allow implicit conversion from C string to std::string. */
   simdjson_result<document_stream> iterate_many(const char *buf, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept = delete;

+#ifndef SIMDJSON_DISABLE_DEPRECATED_API
+  /**
+   * @deprecated Use iterate_many with stream_format::comma_delimited instead.
+   */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(padded_string_view json, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) */
+  simdjson_deprecated inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size, bool allow_comma_separated) noexcept;
+  /** @private Deleted for the same reason as iterate_many(const std::string &&s, size_t batch_size). */
+  inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size, bool allow_comma_separated) = delete;// unsafe
+  /** @private @overload iterate_many(const std::string &&s, size_t batch_size, bool allow_comma_separated) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size, bool allow_comma_separated) = delete;// unsafe
+#endif // SIMDJSON_DISABLE_DEPRECATED_API
+
+  /**
+   * Parse a stream of JSON documents with explicit format specification.
+   *
+   * @param buf The concatenated JSON documents.
+   * @param len The length of the buffer.
+   * @param batch_size The batch size to use.
+   * @param format The stream format (whitespace_delimited, json_sequence, or comma_delimited).
+   * @return A stream of documents, or an error.
+   */
+  inline simdjson_result<document_stream> iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(padded_string_view s, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(const std::string &s, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format)
+   *
+   * The string is padded in place if needed (see simdjson::pad), as with iterate_many(std::string &s, size_t batch_size).
+   */
+  inline simdjson_result<document_stream> iterate_many(std::string &s, size_t batch_size, stream_format format) noexcept;
+  /** @overload iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &s, size_t batch_size, stream_format format) noexcept;
+  /** @private Deleted for the same reason as iterate_many(const std::string &&s, size_t batch_size). */
+  inline simdjson_result<document_stream> iterate_many(const std::string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+  /** @private @overload iterate_many(const std::string &&s, size_t batch_size, stream_format format) */
+  inline simdjson_result<document_stream> iterate_many(const padded_string &&s, size_t batch_size, stream_format format) = delete;// unsafe
+
   /** The capacity of this parser (the largest document it can process). */
   simdjson_pure simdjson_inline size_t capacity() const noexcept;
   /** The maximum capacity of this parser (the largest document it is allowed to process). */
@@ -176116,6 +224882,7 @@ private:
   size_t _capacity{0};
   size_t _max_capacity;
   size_t _max_depth{DEFAULT_MAX_DEPTH};
+  size_t _document_len{0};
   std::unique_ptr<uint8_t[]> string_buf{};

 #if SIMDJSON_DEVELOPMENT_CHECKS
@@ -176178,8 +224945,19 @@ public:
    * Begin array iteration.
    *
    * Part of the std::iterable interface.
+   *
+   * With SIMDJSON_DEVELOPMENT_CHECKS, the iterator locks this array while it is
+   * alive, so that reentrant access (at(), count_elements(), reset(), ...) is
+   * reported as OUT_OF_ORDER_ITERATION.
    */
-  simdjson_inline simdjson_result<array_iterator> begin() noexcept;
+  simdjson_inline simdjson_result<array_iterator> begin() & noexcept;
+  /**
+   * Begin iteration over a temporary array, e.g., `v.get_array().begin()`.
+   *
+   * The iterator does not depend on the array instance and may outlive it, so
+   * it does not lock it.
+   */
+  simdjson_inline simdjson_result<array_iterator> begin() && noexcept;
   /**
    * Sentinel representing the end of the array.
    *
@@ -176310,7 +225088,7 @@ public:
    */
   template <typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out)
-     noexcept(custom_deserializable<T, array> ? nothrow_custom_deserializable<T, array> : true) {
+     noexcept(nothrow_gettable<T, array>) {
     static_assert(custom_deserializable<T, array>);
     return deserialize(*this, out);
   }
@@ -176322,7 +225100,7 @@ public:
    */
   template <typename T>
   simdjson_inline simdjson_result<T> get()
-    noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+    noexcept(nothrow_gettable<T, array>)
   {
     static_assert(std::is_default_constructible<T>::value, "The specified type is not default constructible.");
     T out{};
@@ -176378,6 +225156,10 @@ protected:
    * iter.is_alive() == false indicates iteration is complete.
    */
   value_iterator iter{};
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  bool locked{false};
+  simdjson_inline void set_locked(bool _locked) noexcept;
+#endif

   friend class value;
   friend class document;
@@ -176399,7 +225181,8 @@ public:
   simdjson_inline simdjson_result(error_code error) noexcept; ///< @private
   simdjson_inline simdjson_result() noexcept = default;

-  simdjson_inline simdjson_result<rvv_vls::ondemand::array_iterator> begin() noexcept;
+  simdjson_inline simdjson_result<rvv_vls::ondemand::array_iterator> begin() & noexcept;
+  simdjson_inline simdjson_result<rvv_vls::ondemand::array_iterator> begin() && noexcept;
   simdjson_inline simdjson_result<rvv_vls::ondemand::array_iterator> end() noexcept;
   inline simdjson_result<size_t> count_elements() & noexcept;
   inline simdjson_result<bool> is_empty() & noexcept;
@@ -176419,7 +225202,7 @@ public:
   // TODO: move this code into object-inl.h

   template<typename T>
-  simdjson_inline simdjson_result<T> get() noexcept {
+  simdjson_inline simdjson_result<T> get() noexcept(nothrow_gettable<T, rvv_vls::ondemand::array>) {
     if (error()) { return error(); }
     if constexpr (std::is_same_v<T, rvv_vls::ondemand::array>) {
       return first;
@@ -176427,7 +225210,7 @@ public:
     return first.get<T>();
   }
   template<typename T>
-  simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept {
+  simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept(nothrow_gettable<T, rvv_vls::ondemand::array>) {
     if (error()) { return error(); }
     if constexpr (std::is_same_v<T, rvv_vls::ondemand::array>) {
       out = first;
@@ -176479,6 +225262,15 @@ public:
   /** Create a new, invalid array iterator. */
   simdjson_inline array_iterator() noexcept = default;

+#if SIMDJSON_DEVELOPMENT_CHECKS
+  simdjson_inline ~array_iterator() noexcept;
+
+  simdjson_inline array_iterator(array_iterator&&) noexcept;
+  simdjson_inline array_iterator& operator=(array_iterator&&) noexcept;
+  simdjson_inline array_iterator(const array_iterator&) noexcept;
+  simdjson_inline array_iterator& operator=(const array_iterator&) noexcept;
+#endif
+
   //
   // Iterator interface
   //
@@ -176521,6 +225313,9 @@ public:
 private:
 #if SIMDJSON_DEVELOPMENT_CHECKS
    bool has_been_referenced{false};
+   array* parent{nullptr};
+
+   simdjson_inline array_iterator(const value_iterator &_iter, array* _parent) noexcept;
 #endif
   value_iterator iter{};

@@ -176620,14 +225415,14 @@ public:
   /**
    * Cast this JSON value to an unsigned integer.
    *
-   * @returns A signed 64-bit integer.
+   * @returns A unsigned 64-bit integer.
    * @returns INCORRECT_TYPE If the JSON value is not a 64-bit unsigned integer.
    */
   simdjson_inline simdjson_result<uint64_t> get_uint64() noexcept;
   /**
    * Cast this JSON value (inside string) to an unsigned integer.
    *
-   * @returns A signed 64-bit integer.
+   * @returns A unsigned 64-bit integer.
    * @returns INCORRECT_TYPE If the JSON value is not a 64-bit unsigned integer.
    */
   simdjson_inline simdjson_result<uint64_t> get_uint64_in_string() noexcept;
@@ -176665,6 +225460,46 @@ public:
    * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int32_t.
    */
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  /**
+   * Cast this JSON value to a 16-bit unsigned integer.
+   *
+   * Calls get_uint64() and checks that the result fits in a uint16_t.
+   *
+   * @returns A 16-bit unsigned integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint16_t.
+   */
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  /**
+   * Cast this JSON value to a 16-bit signed integer.
+   *
+   * Calls get_int64() and checks that the result fits in an int16_t.
+   *
+   * @returns A 16-bit signed integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int16_t.
+   */
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  /**
+   * Cast this JSON value to an 8-bit unsigned integer.
+   *
+   * Calls get_uint64() and checks that the result fits in a uint8_t.
+   *
+   * @returns An 8-bit unsigned integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an unsigned integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in a uint8_t.
+   */
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  /**
+   * Cast this JSON value to an 8-bit signed integer.
+   *
+   * Calls get_int64() and checks that the result fits in an int8_t.
+   *
+   * @returns An 8-bit signed integer.
+   * @returns INCORRECT_TYPE If the JSON value is not an integer.
+   * @returns NUMBER_OUT_OF_RANGE If the value does not fit in an int8_t.
+   */
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   /**
    * Cast this JSON value to a double.
    *
@@ -176680,6 +225515,53 @@ public:
    * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
    */
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+
+  /**
+   * Cast this JSON value to a float (binary32).
+   *
+   * The value is rounded directly to binary32: it is the float nearest to the
+   * JSON number. Note that this may differ from get_double() followed by a cast
+   * to float, which rounds twice.
+   *
+   * @returns A float.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+
+  /**
+   * Cast this JSON value (inside string) to a float (binary32).
+   *
+   * @returns A float.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  /**
+   * Cast this JSON value to a std::float32_t (C++23).
+   *
+   * Same as get_float(): the value is rounded directly to binary32.
+   *
+   * @returns A std::float32_t.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   * @returns NUMBER_ERROR If the JSON number is too large in magnitude to be a finite float.
+   */
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  /**
+   * Cast this JSON value to a std::float64_t (C++23).
+   *
+   * Same as get_double().
+   *
+   * @returns A std::float64_t.
+   * @returns INCORRECT_TYPE If the JSON value is not a valid floating-point number.
+   */
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   /**
    * Cast this JSON value to a string.
    *
@@ -176693,6 +225575,24 @@ public:
    * @returns INCORRECT_TYPE if the JSON value is not a string.
    */
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  /**
+   * Cast this JSON value to a C++20 UTF-8 string.
+   *
+   * The string is guaranteed to be valid UTF-8.
+   *
+   * Equivalent to get<std::u8string_view>(). This is a zero-copy alias of
+   * get_string(): the very same bytes are returned, viewed as char8_t.
+   *
+   * Important: Calling get_u8string() twice on the same document is an error.
+   *
+   * @param allow_replacement Whether to allow a replacement character for unmatched surrogate pairs.
+   * @returns An UTF-8 string. The string is stored in the parser and will be invalidated the next
+   *          time it parses a document or when it is destroyed.
+   * @returns INCORRECT_TYPE if the JSON value is not a string.
+   */
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   /**
    * Attempts to fill the provided std::string reference with the parsed value of the current string.
    *
@@ -176763,9 +225663,12 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool
+   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool
    *
-   * You may use get_double(), get_bool(), get_uint64(), get_int64(),
+   * You may use get_double(), get_float(), get_bool(), get_uint64(), get_int64(),
+   * get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(),
+   * get_float32(), get_float64(),
    * get_object(), get_array(), get_raw_json_string(), or get_string() instead.
    *
    * @returns A value of the given type, parsed from the JSON.
@@ -176774,7 +225677,7 @@ public:
   template <typename T>
   simdjson_inline simdjson_result<T> get() &
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    noexcept(nothrow_gettable<T, document>)
 #else
     noexcept
 #endif
@@ -176797,7 +225700,7 @@ public:
   template<typename T>
   simdjson_inline simdjson_result<T> get() &&
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    noexcept(nothrow_gettable<T, document>)
 #else
     noexcept
 #endif
@@ -176809,7 +225712,8 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool, value
+   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool, value
    *
    * Be mindful that the document instance must remain in scope while you are accessing object, array and value instances.
    *
@@ -176820,7 +225724,7 @@ public:
   template<typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out) &
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    noexcept(nothrow_gettable<T, document>)
 #else
     noexcept
 #endif
@@ -176833,7 +225737,7 @@ public:
         "And you do not seem to have added support for it. Indeed, we have that "
         "simdjson::custom_deserializable<T> is false and the type T is not a default type "
         "such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-        "int64_t, double, or bool.");
+        "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
       static_cast<void>(out); // to get rid of unused errors
       return UNINITIALIZED;
     }
@@ -176842,7 +225746,8 @@ public:
     // immediately fail.
     static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
       "The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-      "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+      "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+      "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
       " get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
       " You may also add support for custom types, see our documentation.");
     static_cast<void>(out); // to get rid of unused errors
@@ -176851,7 +225756,12 @@ public:
   }

   /** @overload template<typename T> error_code get(T &out) & noexcept */
-  template<typename T> simdjson_deprecated simdjson_inline error_code get(T &out) && noexcept;
+  template<typename T> simdjson_deprecated simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, document>);
+#else
+    noexcept;
+#endif

 #if SIMDJSON_EXCEPTIONS
   /**
@@ -176885,24 +225795,24 @@ public:
   /**
    * Cast this JSON value to an unsigned integer.
    *
-   * @returns A signed 64-bit integer.
+   * @returns A unsigned 64-bit integer.
    * @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit unsigned integer.
    */
-  simdjson_inline operator uint64_t() noexcept(false);
+  explicit simdjson_inline operator uint64_t() noexcept(false);
   /**
    * Cast this JSON value to a signed integer.
    *
    * @returns A signed 64-bit integer.
    * @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a 64-bit integer.
    */
-  simdjson_inline operator int64_t() noexcept(false);
+  explicit simdjson_inline operator int64_t() noexcept(false);
   /**
    * Cast this JSON value to a double.
    *
    * @returns A double.
    * @exception simdjson_error(INCORRECT_TYPE) If the JSON value is not a valid floating-point number.
    */
-  simdjson_inline operator double() noexcept(false);
+  explicit simdjson_inline operator double() noexcept(false);
   /**
    * Cast this JSON value to a string.
    *
@@ -176912,7 +225822,7 @@ public:
    *          time it parses a document or when it is destroyed.
    * @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not a string.
    */
-  simdjson_inline operator std::string_view() noexcept(false) simdjson_lifetime_bound;
+  explicit simdjson_inline operator std::string_view() noexcept(false) simdjson_lifetime_bound;
   /**
    * Cast this JSON value to a raw_json_string.
    *
@@ -176921,14 +225831,14 @@ public:
    * @returns A pointer to the raw JSON for the given string.
    * @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not a string.
    */
-  simdjson_inline operator raw_json_string() noexcept(false) simdjson_lifetime_bound;
+  explicit simdjson_inline operator raw_json_string() noexcept(false) simdjson_lifetime_bound;
   /**
    * Cast this JSON value to a bool.
    *
    * @returns A bool value.
    * @exception simdjson_error(INCORRECT_TYPE) if the JSON value is not true or false.
    */
-  simdjson_inline operator bool() noexcept(false);
+  explicit simdjson_inline operator bool() noexcept(false);
   /**
    * Cast this JSON value to a value when the document is an object or an array.
    *
@@ -177423,9 +226333,24 @@ public:
   simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template <typename string_type>
   simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -177437,7 +226362,7 @@ public:
   template <typename T>
   simdjson_inline simdjson_result<T> get() &
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    noexcept(nothrow_gettable<T, document_reference>)
 #else
     noexcept
 #endif
@@ -177450,7 +226375,8 @@ public:
   template<typename T>
   simdjson_inline simdjson_result<T> get() &&
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document> : true)
+    // Forwards to document::get<T>(), so the document customization decides.
+    noexcept(nothrow_gettable<T, document>)
 #else
     noexcept
 #endif
@@ -177462,7 +226388,8 @@ public:
   /**
    * Get this value as the given type.
    *
-   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, double, bool, value
+   * Supported types: object, array, raw_json_string, string_view, uint64_t, int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float,
+   * std::float64_t and std::float32_t (C++23, when available), bool, value
    *
    * Be mindful that the document instance must remain in scope while you are accessing object, array and value instances.
    *
@@ -177473,7 +226400,7 @@ public:
   template<typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out) &
 #if SIMDJSON_SUPPORTS_CONCEPTS
-    noexcept(custom_deserializable<T, document> ? nothrow_custom_deserializable<T, document_reference> : true)
+    noexcept(nothrow_gettable<T, document_reference>)
 #else
     noexcept
 #endif
@@ -177486,7 +226413,7 @@ public:
         "And you do not seem to have added support for it. Indeed, we have that "
         "simdjson::custom_deserializable<T> is false and the type T is not a default type "
         "such as ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-        "int64_t, double, or bool.");
+        "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, or bool.");
       static_cast<void>(out); // to get rid of unused errors
       return UNINITIALIZED;
     }
@@ -177495,7 +226422,8 @@ public:
     // immediately fail.
     static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
       "The supported types are ondemand::object, ondemand::array, raw_json_string, std::string_view, uint64_t, "
-      "int64_t, double, and bool. We recommend you use get_double(), get_bool(), get_uint64(), get_int64(), "
+      "int64_t, uint32_t, int32_t, uint16_t, int16_t, uint8_t, int8_t, double, float, and bool. We recommend you use get_double(), get_float(), "
+      "get_bool(), get_uint64(), get_int64(), get_uint32(), get_int32(), get_uint16(), get_int16(), get_uint8(), get_int8(), "
       " get_object(), get_array(), get_raw_json_string(), or get_string() instead of the get template."
       " You may also add support for custom types, see our documentation.");
     static_cast<void>(out); // to get rid of unused errors
@@ -177504,7 +226432,12 @@ public:
   }

   /** @overload template<typename T> error_code get(T &out) & noexcept */
-  template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+  template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, document_reference>);
+#else
+    noexcept;
+#endif
   simdjson_inline simdjson_result<std::string_view> raw_json() noexcept;
 #if SIMDJSON_STATIC_REFLECTION
   template<constevalutil::fixed_string... FieldNames, typename T>
@@ -177517,12 +226450,12 @@ public:
   explicit simdjson_inline operator T() noexcept(false);
   simdjson_inline operator array() & noexcept(false);
   simdjson_inline operator object() & noexcept(false);
-  simdjson_inline operator uint64_t() noexcept(false);
-  simdjson_inline operator int64_t() noexcept(false);
-  simdjson_inline operator double() noexcept(false);
-  simdjson_inline operator std::string_view() noexcept(false);
-  simdjson_inline operator raw_json_string() noexcept(false);
-  simdjson_inline operator bool() noexcept(false);
+  explicit simdjson_inline operator uint64_t() noexcept(false);
+  explicit simdjson_inline operator int64_t() noexcept(false);
+  explicit simdjson_inline operator double() noexcept(false);
+  explicit simdjson_inline operator std::string_view() noexcept(false);
+  explicit simdjson_inline operator raw_json_string() noexcept(false);
+  explicit simdjson_inline operator bool() noexcept(false);
   simdjson_inline operator value() noexcept(false);
 #endif
   simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -177584,9 +226517,24 @@ public:
   simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template <typename string_type>
   simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -177595,11 +226543,31 @@ public:
   simdjson_inline simdjson_result<rvv_vls::ondemand::value> get_value() noexcept;
   simdjson_inline simdjson_result<bool> is_null() noexcept;

-  template<typename T> simdjson_inline simdjson_result<T> get() & noexcept;
-  template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() && noexcept;
+  template<typename T> simdjson_inline simdjson_result<T> get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, rvv_vls::ondemand::document>);
+#else
+    noexcept;
+#endif
+  template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, rvv_vls::ondemand::document>);
+#else
+    noexcept;
+#endif

-  template<typename T> simdjson_inline error_code get(T &out) & noexcept;
-  template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+  template<typename T> simdjson_inline error_code get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, rvv_vls::ondemand::document>);
+#else
+    noexcept;
+#endif
+  template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, rvv_vls::ondemand::document>);
+#else
+    noexcept;
+#endif
 #if SIMDJSON_EXCEPTIONS

   using rvv_vls::implementation_simdjson_result_base<rvv_vls::ondemand::document>::operator*;
@@ -177608,12 +226576,12 @@ public:
   explicit simdjson_inline operator T() noexcept(false);
   simdjson_inline operator rvv_vls::ondemand::array() & noexcept(false);
   simdjson_inline operator rvv_vls::ondemand::object() & noexcept(false);
-  simdjson_inline operator uint64_t() noexcept(false);
-  simdjson_inline operator int64_t() noexcept(false);
-  simdjson_inline operator double() noexcept(false);
-  simdjson_inline operator std::string_view() noexcept(false);
-  simdjson_inline operator rvv_vls::ondemand::raw_json_string() noexcept(false);
-  simdjson_inline operator bool() noexcept(false);
+  explicit simdjson_inline operator uint64_t() noexcept(false);
+  explicit simdjson_inline operator int64_t() noexcept(false);
+  explicit simdjson_inline operator double() noexcept(false);
+  explicit simdjson_inline operator std::string_view() noexcept(false);
+  explicit simdjson_inline operator rvv_vls::ondemand::raw_json_string() noexcept(false);
+  explicit simdjson_inline operator bool() noexcept(false);
   simdjson_inline operator rvv_vls::ondemand::value() noexcept(false);
 #endif
   simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -177679,9 +226647,24 @@ public:
   simdjson_inline simdjson_result<int64_t> get_int64_in_string() noexcept;
   simdjson_inline simdjson_result<uint32_t> get_uint32() noexcept;
   simdjson_inline simdjson_result<int32_t> get_int32() noexcept;
+  simdjson_inline simdjson_result<uint16_t> get_uint16() noexcept;
+  simdjson_inline simdjson_result<int16_t> get_int16() noexcept;
+  simdjson_inline simdjson_result<uint8_t> get_uint8() noexcept;
+  simdjson_inline simdjson_result<int8_t> get_int8() noexcept;
   simdjson_inline simdjson_result<double> get_double() noexcept;
   simdjson_inline simdjson_result<double> get_double_in_string() noexcept;
+  simdjson_inline simdjson_result<float> get_float() noexcept;
+  simdjson_inline simdjson_result<float> get_float_in_string() noexcept;
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  simdjson_inline simdjson_result<std::float32_t> get_float32() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+  simdjson_inline simdjson_result<std::float64_t> get_float64() noexcept;
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
   simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> get_u8string(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template <typename string_type>
   simdjson_warn_unused simdjson_inline error_code get_string(string_type& receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<std::string_view> get_wobbly_string() noexcept;
@@ -177690,22 +226673,42 @@ public:
   simdjson_inline simdjson_result<rvv_vls::ondemand::value> get_value() noexcept;
   simdjson_inline simdjson_result<bool> is_null() noexcept;

-  template<typename T> simdjson_inline simdjson_result<T> get() & noexcept;
-  template<typename T> simdjson_inline simdjson_result<T> get() && noexcept;
+  template<typename T> simdjson_inline simdjson_result<T> get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, rvv_vls::ondemand::document_reference>);
+#else
+    noexcept;
+#endif
+  template<typename T> simdjson_inline simdjson_result<T> get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, rvv_vls::ondemand::document_reference>);
+#else
+    noexcept;
+#endif

-  template<typename T> simdjson_inline error_code get(T &out) & noexcept;
-  template<typename T> simdjson_inline error_code get(T &out) && noexcept;
+  template<typename T> simdjson_inline error_code get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, rvv_vls::ondemand::document_reference>);
+#else
+    noexcept;
+#endif
+  template<typename T> simdjson_inline error_code get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, rvv_vls::ondemand::document_reference>);
+#else
+    noexcept;
+#endif
 #if SIMDJSON_EXCEPTIONS
   template <class T>
   explicit simdjson_inline operator T() noexcept(false);
   simdjson_inline operator rvv_vls::ondemand::array() & noexcept(false);
   simdjson_inline operator rvv_vls::ondemand::object() & noexcept(false);
-  simdjson_inline operator uint64_t() noexcept(false);
-  simdjson_inline operator int64_t() noexcept(false);
-  simdjson_inline operator double() noexcept(false);
-  simdjson_inline operator std::string_view() noexcept(false);
-  simdjson_inline operator rvv_vls::ondemand::raw_json_string() noexcept(false);
-  simdjson_inline operator bool() noexcept(false);
+  explicit simdjson_inline operator uint64_t() noexcept(false);
+  explicit simdjson_inline operator int64_t() noexcept(false);
+  explicit simdjson_inline operator double() noexcept(false);
+  explicit simdjson_inline operator std::string_view() noexcept(false);
+  explicit simdjson_inline operator rvv_vls::ondemand::raw_json_string() noexcept(false);
+  explicit simdjson_inline operator bool() noexcept(false);
   simdjson_inline operator rvv_vls::ondemand::value() noexcept(false);
 #endif
   simdjson_inline simdjson_result<size_t> count_elements() & noexcept;
@@ -177873,10 +226876,7 @@ public:
    *   }
    *   size_t truncated = stream.truncated_bytes();
    *
-   * IMPORTANT: this value is only meaningful under the conditions below. It is
-   * computed from stage-1 bookkeeping, and outside these conditions it is not
-   * merely imprecise, it is arbitrary -- it can exceed size_in_bytes() or wrap
-   * around to a huge value. Check it only when both of the following hold:
+   * IMPORTANT: this value is only meaningful under the conditions below.
    *
    *   - you iterated all the way to the end of the stream;
    *   - no document reported an error. Iteration stops at the first failed
@@ -177885,6 +226885,9 @@ public:
    * If you need to know about a truncated tail outside those conditions, track
    * it yourself from the last successful document (see iterator::current_index()
    * and iterator::source()).
+   *
+   * An empty input (zero bytes) or an input made only of white space contains
+   * no document: truncated_bytes() returns zero.
    */
   inline size_t truncated_bytes() const noexcept;

@@ -177944,7 +226947,10 @@ public:
      *
      * The returned string_view instance is simply a map to the (unparsed)
      * source string: it may thus include white-space characters and all manner
-     * of padding.
+     * of padding. It spans the whole current document, whether or not you
+     * have already accessed (part of) the document. Thus
+     * current_index() + source().size() is the offset just past the end of the
+     * current document, which is useful when reading a stream in chunks.
      *
      * This function (source()) is experimental and the usage
      * may change in future versions of simdjson: we find the API somewhat
@@ -177998,13 +227004,16 @@ private:
    * @param buf is the raw byte buffer we need to process
    * @param len is the length of the raw byte buffer in bytes
    * @param batch_size is the size of the windows (must be strictly greater or equal to the largest JSON document)
+   * @param allow_comma_separated whether to allow comma-separated documents
+   * @param format the stream format
    */
   simdjson_inline document_stream(
     ondemand::parser &parser,
     const uint8_t *buf,
     size_t len,
     size_t batch_size,
-    bool allow_comma_separated
+    bool allow_comma_separated,
+    stream_format format = stream_format::whitespace_delimited
   ) noexcept;

   /**
@@ -178038,8 +227047,23 @@ private:
    */
   inline void next() noexcept;

-  /** Move the json_iterator of the document to the location of the next document in the stream. */
+  /**
+   * Move the json_iterator of the document to the location of the next document
+   * in the stream.
+   *
+   * For formats with a document delimiter (`newline_delimited`, `json_sequence`),
+   * when the iterator is still inside the current document (`depth() > 0`), this
+   * may jump to the next delimiter instead of walking remaining structurals. That
+   * jump does not structure-validate the unread remainder.
+   */
   inline void next_document() noexcept;
+  /** Byte that ends a document under `format`, or 0 if there is none. */
+  simdjson_inline uint8_t document_delimiter() const noexcept;
+  /**
+   * Position the iterator at the first structural at or past the next
+   * `delimiter` in the current batch. Returns false if none is found.
+   */
+  simdjson_inline bool skip_to_delimiter(uint8_t delimiter) noexcept;

   /** Get the next document index. */
   inline size_t next_batch_start() const noexcept;
@@ -178053,6 +227077,7 @@ private:
   size_t len;
   size_t batch_size;
   bool allow_comma_separated;
+  stream_format format;
   /**
    * We are going to use just one document instance. The document owns
    * the json_iterator. It implies that we only ever pass a reference
@@ -178079,7 +227104,7 @@ private:
   /** The error returned from the stage 1 thread. */
   error_code stage1_thread_error{UNINITIALIZED};
   /** The thread used to run stage 1 against the next batch in the background. */
-  std::unique_ptr<stage1_worker> worker{new(std::nothrow) stage1_worker()};
+  std::unique_ptr<stage1_worker> worker{};
   /**
    * The parser used to run stage 1 in the background. Will be swapped
    * with the regular parser when finished.
@@ -178154,6 +227179,16 @@ public:
    * call it again nor can you call key().
    */
   simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  /**
+   * Get the key as a C++20 u8string_view. This is a zero-copy alias of
+   * unescaped_key(): the very same bytes are returned, viewed as char8_t.
+   *
+   * This consumes the key: once you have called unescaped_u8key(), you cannot
+   * call it again nor can you call key().
+   */
+  simdjson_inline simdjson_warn_unused simdjson_result<std::u8string_view> unescaped_u8key(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   /**
    * Get the key as a string_view (for higher speed, consider raw_key).
    * We deliberately use a more cumbersome name (unescaped_key) to force users
@@ -178186,6 +227221,16 @@ public:
    * you can safely call it repeatedly. Use unescaped_key() to get the unescaped key.
    */
   simdjson_inline std::string_view escaped_key() const noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  /**
+   * Get the key as a C++20 u8string_view. This is a zero-copy alias of
+   * escaped_key(): the very same bytes are returned, viewed as char8_t.
+   * The string is unprocessed, so it may contain escape characters
+   * (e.g., \uXXXX or \n). It does not count as a consumption of the content:
+   * you can safely call it repeatedly.
+   */
+  simdjson_inline std::u8string_view escaped_u8key() const noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   /**
    * Get the field value.
    */
@@ -178217,11 +227262,17 @@ public:
   simdjson_inline simdjson_result() noexcept = default;

   simdjson_inline simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> unescaped_u8key(bool allow_replacement = false) noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   template<typename string_type>
   simdjson_inline error_code unescaped_key(string_type &receiver, bool allow_replacement = false) noexcept;
   simdjson_inline simdjson_result<rvv_vls::ondemand::raw_json_string> key() noexcept;
   simdjson_inline simdjson_result<std::string_view> key_raw_json_token() noexcept;
   simdjson_inline simdjson_result<std::string_view> escaped_key() noexcept;
+#if SIMDJSON_SUPPORTS_CHAR8_T
+  simdjson_inline simdjson_result<std::u8string_view> escaped_u8key() noexcept;
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
   simdjson_inline simdjson_result<rvv_vls::ondemand::value> value() noexcept;
 };

@@ -178229,6 +227280,1398 @@ public:

 #endif // SIMDJSON_GENERIC_ONDEMAND_FIELD_H
 /* end file simdjson/generic/ondemand/field.h for rvv_vls */
+/* including simdjson/generic/ondemand/key_selector.h for rvv_vls: #include "simdjson/generic/ondemand/key_selector.h" */
+/* begin file simdjson/generic/ondemand/key_selector.h for rvv_vls */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+#define SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #include "simdjson/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/common_defs.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/constevalutil.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/raw_json_string.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#include <array>
+#include <string>      // std::string (key_selector::describe)
+#include <string_view>
+#include <cstddef>
+#include <cstdint>
+#include <cstring>     // std::memcpy (portable unaligned window load)
+#include <utility>     // std::index_sequence (window candidate dispatch)
+#include <type_traits> // std::integral_constant (window candidate dispatch)
+
+// AArch64 only: 32-bit ARM (ARMv7) defines __ARM_NEON too, but lacks the
+// A64-only horizontal reductions (vmaxvq_u32) used below. See issue #2885.
+#if defined(__aarch64__)
+  #include <arm_neon.h>
+  #define SIMDJSON_KEY_SELECTOR_HAS_NEON 1
+#else
+  #define SIMDJSON_KEY_SELECTOR_HAS_NEON 0
+#endif
+#if defined(__SSE2__)
+  #include <emmintrin.h>
+  #define SIMDJSON_KEY_SELECTOR_HAS_SSE2 1
+#else
+  #define SIMDJSON_KEY_SELECTOR_HAS_SSE2 0
+#endif
+#if defined(__loongarch_sx)
+  #include <lsxintrin.h>
+  #define SIMDJSON_KEY_SELECTOR_HAS_LSX 1
+#else
+  #define SIMDJSON_KEY_SELECTOR_HAS_LSX 0
+#endif
+
+#if SIMDJSON_SUPPORTS_CONCEPTS
+
+namespace simdjson {
+namespace rvv_vls {
+namespace ondemand {
+
+namespace key_selector_detail {
+
+// Since not constexpr, triggers compile-time error.
+inline void compile_time_error(const char* message) noexcept { (void)message; }
+
+// ============================================================================
+// Compile-time perfect-hash generator.
+// It scales to ~100 keys at compile time by determining association values one (position, character)
+// symbol at a time (gperf-style) instead of an exhaustive offset search, and
+// falls back to a Hash-and-Displace construction for large/awkward key sets.
+//
+// Only flat tables survive to runtime; the lookup is a few additions plus a
+// single SIMD key comparison (see match_raw below).
+// ============================================================================
+
+// Maximum number of character positions the gperf hash may combine.
+static constexpr std::size_t MAX_POSITIONS = 16;
+// Sentinel "position" meaning "the last character of the key".
+static constexpr std::size_t LAST_CHAR = std::size_t(-1);
+// Runtime-encoded sentinels (stored in uint8 tables).
+static constexpr std::uint8_t POS_LAST_CHAR = 0xFF; // positions_[i] == last char
+static constexpr std::uint8_t HD_MODE       = 0xFF; // num_positions == H&D mode
+// Flags stored in positions[2] in H&D mode to select the key-hash variant.
+static constexpr std::size_t HD_HASH_2BYTE_FLAG = 2;
+static constexpr std::size_t HD_HASH_4BYTE_FLAG = 4;
+
+constexpr std::size_t next_power_of_2(std::size_t n) noexcept {
+    if (n == 0) { return 1; }
+    std::size_t p = 1;
+    while (p < n) { p <<= 1; }
+    return p;
+}
+
+// Character at a given position (LAST_CHAR means last character), or 256 if out
+// of bounds.
+constexpr std::size_t char_at(std::string_view key, std::size_t pos) noexcept {
+    if (pos == LAST_CHAR) {
+        if (key.empty()) { return 256; }
+        return static_cast<unsigned char>(key[key.size() - 1]);
+    }
+    if (pos >= key.size()) { return 256; }
+    return static_cast<unsigned char>(key[pos]);
+}
+
+// Count key pairs that a set of positions fails to distinguish. Keys whose
+// lengths differ modulo the table size are separated by the length term in the
+// hash, so they need no position coverage.
+template <std::size_t N>
+consteval std::size_t count_undistinguished_pairs(
+    const std::array<std::string_view, N>& keys,
+    const std::size_t* positions,
+    std::size_t num_positions,
+    std::size_t modulus) {
+    std::size_t count = 0;
+    for (std::size_t i = 0; i < N; ++i) {
+        for (std::size_t j = i + 1; j < N; ++j) {
+            if (keys[i].size() % modulus != keys[j].size() % modulus) { continue; }
+            bool distinguished = false;
+            for (std::size_t p = 0; p < num_positions; ++p) {
+                if (char_at(keys[i], positions[p]) != char_at(keys[j], positions[p])) {
+                    distinguished = true;
+                    break;
+                }
+            }
+            if (!distinguished) { ++count; }
+        }
+    }
+    return count;
+}
+
+template <std::size_t N>
+consteval bool positions_distinguish(
+    const std::array<std::string_view, N>& keys,
+    const std::size_t* positions,
+    std::size_t num_positions,
+    std::size_t modulus) {
+    return count_undistinguished_pairs<N>(keys, positions, num_positions, modulus) == 0;
+}
+
+// Number of distinct (length % modulus, char_at(key, pos)) pairs at a position.
+template <std::size_t N>
+consteval std::size_t discriminating_power(
+    const std::array<std::string_view, N>& keys,
+    std::size_t pos,
+    std::size_t modulus) {
+    struct pair { std::size_t len_mod; std::size_t ch; };
+    std::array<pair, N> pairs{};
+    for (std::size_t i = 0; i < N; ++i) {
+        pairs[i] = {keys[i].size() % modulus, char_at(keys[i], pos)};
+    }
+    std::size_t count = 0;
+    for (std::size_t i = 0; i < N; ++i) {
+        bool dup = false;
+        for (std::size_t j = 0; j < i; ++j) {
+            if (pairs[i].len_mod == pairs[j].len_mod && pairs[i].ch == pairs[j].ch) {
+                dup = true;
+                break;
+            }
+        }
+        if (!dup) { ++count; }
+    }
+    return count;
+}
+
+template <std::size_t N>
+consteval std::size_t max_key_length(const std::array<std::string_view, N>& keys) {
+    std::size_t m = 0;
+    for (std::size_t i = 0; i < N; ++i) {
+        if (keys[i].size() > m) { m = keys[i].size(); }
+    }
+    return m;
+}
+
+// Bounded backtracking DFS for a minimal set of distinguishing positions.
+template <std::size_t N>
+consteval bool backtracking_search(
+    const std::array<std::string_view, N>& keys,
+    const std::size_t* candidates,
+    std::size_t num_candidates,
+    std::size_t* positions,
+    std::size_t& num_positions_out,
+    std::size_t& budget,
+    std::size_t modulus) {
+    constexpr std::size_t MAX_DEPTH = 8;
+    std::size_t breadth = num_candidates < 20 ? num_candidates : 20;
+
+    struct frame { std::size_t depth; std::size_t next_ci; std::size_t parent_count; };
+    std::array<frame, MAX_DEPTH + 1> stack{};
+    std::size_t sp = 0;
+
+    std::size_t initial_count = count_undistinguished_pairs<N>(keys, positions, 0, modulus);
+    if (budget > 0) { --budget; }
+    if (initial_count == 0) { num_positions_out = 0; return true; }
+
+    stack[0] = {0, 0, initial_count};
+
+    while (budget > 0) {
+        if (sp > MAX_DEPTH) {
+            if (sp == 0) { break; }
+            --sp;
+            ++stack[sp].next_ci;
+            continue;
+        }
+        auto& f = stack[sp];
+        if (f.next_ci >= breadth) {
+            if (sp == 0) { break; }
+            --sp;
+            ++stack[sp].next_ci;
+            continue;
+        }
+        positions[sp] = candidates[f.next_ci];
+        --budget;
+        std::size_t new_count = count_undistinguished_pairs<N>(keys, positions, sp + 1, modulus);
+        if (new_count == 0) { num_positions_out = sp + 1; return true; }
+        if (new_count < f.parent_count && sp + 1 < MAX_DEPTH) {
+            stack[sp + 1] = {sp + 1, f.next_ci + 1, new_count};
+            ++sp;
+        } else {
+            ++f.next_ci;
+        }
+    }
+    return false;
+}
+
+// Phase 1: select character positions that distinguish all colliding pairs.
+template <std::size_t N>
+consteval std::size_t select_positions(
+    const std::array<std::string_view, N>& keys,
+    std::array<std::size_t, MAX_POSITIONS>& positions,
+    std::size_t modulus) {
+    if (positions_distinguish<N>(keys, positions.data(), 0, modulus)) { return 0; }
+
+    std::size_t max_len = max_key_length(keys);
+    constexpr std::size_t MAX_CANDIDATES = 256;
+    std::array<std::size_t, MAX_CANDIDATES> candidates{};
+    std::array<std::size_t, MAX_CANDIDATES> powers{};
+    std::size_t num_candidates = 0;
+    for (std::size_t p = 0; p < max_len && num_candidates < MAX_CANDIDATES - 1; ++p) {
+        candidates[num_candidates] = p;
+        powers[num_candidates] = discriminating_power(keys, p, modulus);
+        ++num_candidates;
+    }
+    if (num_candidates < MAX_CANDIDATES) {
+        candidates[num_candidates] = LAST_CHAR;
+        powers[num_candidates] = discriminating_power(keys, LAST_CHAR, modulus);
+        ++num_candidates;
+    }
+    for (std::size_t i = 0; i < num_candidates; ++i) {
+        for (std::size_t j = i + 1; j < num_candidates; ++j) {
+            if (powers[j] > powers[i]) {
+                auto tc = candidates[i]; candidates[i] = candidates[j]; candidates[j] = tc;
+                auto tp = powers[i]; powers[i] = powers[j]; powers[j] = tp;
+            }
+        }
+    }
+
+    positions[0] = candidates[0];
+    if (positions_distinguish<N>(keys, positions.data(), 1, modulus)) { return 1; }
+
+    positions[0] = 0;
+    positions[1] = LAST_CHAR;
+    if (positions_distinguish<N>(keys, positions.data(), 2, modulus)) { return 2; }
+
+    {
+        std::size_t budget = 5000;
+        std::size_t num_found = 0;
+        if (backtracking_search<N>(keys, candidates.data(), num_candidates,
+                                   positions.data(), num_found, budget, modulus)) {
+            return num_found;
+        }
+    }
+
+    std::size_t num_pos = 0;
+    for (std::size_t ci = 0; ci < num_candidates && num_pos < MAX_POSITIONS; ++ci) {
+        bool already = false;
+        for (std::size_t p = 0; p < num_pos; ++p) {
+            if (positions[p] == candidates[ci]) { already = true; break; }
+        }
+        if (already) { continue; }
+        positions[num_pos] = candidates[ci];
+        ++num_pos;
+        if (positions_distinguish<N>(keys, positions.data(), num_pos, modulus)) { return num_pos; }
+    }
+
+    compile_time_error("Failed to find distinguishing positions for perfect hash");
+    return 0;
+}
+
+// Result of PHF computation. A max-sized slot_to_key array lets the same struct
+// type carry any chosen table size.
+template <std::size_t N>
+struct phf_result {
+    // Allow up to 8x the minimum table size. Sparser tables solve faster.
+    static constexpr std::size_t MAX_TABLE_SIZE = next_power_of_2(N) * 8;
+    std::size_t table_size{};
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso_values{};
+    std::size_t num_positions{};
+    std::array<std::size_t, MAX_POSITIONS> positions{};
+    std::array<std::size_t, MAX_TABLE_SIZE> slot_to_key{};
+};
+
+// Partition-based asso_values search (gperf-style). Determines asso_values one
+// (position, character) symbol at a time; never revisits a value. Equivalence
+// classes (keys sharing the same undetermined symbols) keep the search cheap.
+template <std::size_t N, std::size_t M>
+consteval bool try_generate_gperf(
+    const std::array<std::string_view, N>& keys,
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS>& asso_values,
+    std::size_t& num_positions,
+    std::array<std::size_t, MAX_POSITIONS>& positions,
+    std::array<std::size_t, M>& slot_to_key) {
+    num_positions = select_positions<N>(keys, positions, M);
+
+    for (std::size_t p = 0; p < MAX_POSITIONS; ++p) {
+        for (std::size_t c = 0; c < 256; ++c) { asso_values[p][c] = 0; }
+    }
+
+    if (num_positions == 0) {
+        for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+        for (std::size_t i = 0; i < N; ++i) {
+            std::size_t slot = keys[i].size() % M;
+            if (slot_to_key[slot] != N) { return false; }
+            slot_to_key[slot] = i;
+        }
+        return true;
+    }
+
+    std::array<std::array<std::size_t, MAX_POSITIONS>, N> kchars{};
+    for (std::size_t k = 0; k < N; ++k) {
+        for (std::size_t p = 0; p < num_positions; ++p) {
+            kchars[k][p] = char_at(keys[k], positions[p]);
+        }
+    }
+
+    struct sym_t { std::size_t pos; std::size_t ch; std::size_t freq; };
+    constexpr std::size_t MAX_SYMS = MAX_POSITIONS * 256;
+    std::array<sym_t, MAX_SYMS> syms{};
+    std::size_t nsyms = 0;
+    for (std::size_t p = 0; p < num_positions; ++p) {
+        std::array<std::size_t, 256> freq{};
+        for (std::size_t k = 0; k < N; ++k) {
+            std::size_t c = kchars[k][p];
+            if (c < 256) { freq[c]++; }
+        }
+        for (std::size_t c = 0; c < 256; ++c) {
+            if (freq[c] > 0) { syms[nsyms++] = {p, c, freq[c]}; }
+        }
+    }
+    for (std::size_t i = 0; i < nsyms; ++i) {
+        for (std::size_t j = i + 1; j < nsyms; ++j) {
+            if (syms[j].freq > syms[i].freq) {
+                auto tmp = syms[i]; syms[i] = syms[j]; syms[j] = tmp;
+            }
+        }
+    }
+
+    std::array<std::size_t, N> phash{};
+    for (std::size_t k = 0; k < N; ++k) { phash[k] = keys[k].size(); }
+
+    std::array<std::array<uint64_t, 256>, MAX_POSITIONS> salt{};
+    {
+        uint64_t s = 0x9e3779b97f4a7c15ULL;
+        for (std::size_t p = 0; p < num_positions; ++p) {
+            for (std::size_t c = 0; c < 256; ++c) {
+                s = s * 6364136223846793005ULL + 1442695040888963407ULL;
+                salt[p][c] = s;
+            }
+        }
+    }
+    std::array<uint64_t, N> sig{};
+    for (std::size_t k = 0; k < N; ++k) {
+        uint64_t s = 0;
+        for (std::size_t p = 0; p < num_positions; ++p) {
+            std::size_t c = kchars[k][p];
+            if (c < 256) { s ^= salt[p][c]; }
+        }
+        sig[k] = s;
+    }
+    std::array<std::size_t, N> order{};
+    for (std::size_t k = 0; k < N; ++k) { order[k] = k; }
+
+    std::array<std::size_t, M> slot_gen{};
+    std::size_t gen = 0;
+
+    std::size_t search_limit = next_power_of_2(M);
+    if (search_limit < 32) { search_limit = 32; }
+
+    for (std::size_t si = 0; si < nsyms; ++si) {
+        std::size_t sp = syms[si].pos;
+        std::size_t sc = syms[si].ch;
+
+        uint64_t sp_salt = salt[sp][sc];
+        for (std::size_t k = 0; k < N; ++k) {
+            if (kchars[k][sp] == sc) { sig[k] ^= sp_salt; }
+        }
+
+        for (std::size_t i = 1; i < N; ++i) {
+            std::size_t x = order[i];
+            uint64_t xs = sig[x];
+            std::size_t j = i;
+            while (j > 0 && sig[order[j - 1]] > xs) {
+                order[j] = order[j - 1];
+                --j;
+            }
+            order[j] = x;
+        }
+
+        bool found = false;
+        for (std::size_t v = 0; v < search_limit && !found; ++v) {
+            bool collision = false;
+            std::size_t ci = 0;
+            while (ci < N && !collision) {
+                uint64_t class_sig = sig[order[ci]];
+                std::size_t cj = ci;
+                while (cj < N && sig[order[cj]] == class_sig) { ++cj; }
+                if (cj - ci > 1) {
+                    ++gen;
+                    for (std::size_t x = ci; x < cj; ++x) {
+                        std::size_t k = order[x];
+                        std::size_t h = phash[k];
+                        if (kchars[k][sp] == sc) { h += v; }
+                        h %= M;
+                        if (slot_gen[h] == gen) { collision = true; break; }
+                        slot_gen[h] = gen;
+                    }
+                }
+                ci = cj;
+            }
+            if (!collision) {
+                asso_values[sp][sc] = v;
+                for (std::size_t k = 0; k < N; ++k) {
+                    if (kchars[k][sp] == sc) { phash[k] += v; }
+                }
+                found = true;
+            }
+        }
+        if (!found) { return false; }
+    }
+
+    for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+    for (std::size_t i = 0; i < N; ++i) {
+        std::size_t slot = phash[i] % M;
+        if (slot_to_key[slot] != N) { return false; }
+        slot_to_key[slot] = i;
+    }
+    std::size_t filled = 0;
+    for (std::size_t i = 0; i < M; ++i) {
+        if (slot_to_key[i] != N) { ++filled; }
+    }
+    return filled == N;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_compute_phf(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+    static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso{};
+    std::size_t npos{};
+    std::array<std::size_t, MAX_POSITIONS> pos{};
+    std::array<std::size_t, M> s2k{};
+    if (try_generate_gperf<N, M>(keys, asso, npos, pos, s2k)) {
+        result.table_size = M;
+        result.asso_values = asso;
+        result.num_positions = npos;
+        result.positions = pos;
+        for (std::size_t i = 0; i < M; ++i) { result.slot_to_key[i] = s2k[i]; }
+        for (std::size_t i = M; i < phf_result<N>::MAX_TABLE_SIZE; ++i) { result.slot_to_key[i] = N; }
+        return true;
+    }
+    return false;
+}
+
+template <std::size_t N, std::size_t M, std::size_t MaxM>
+consteval bool try_gperf_po2(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+    if (try_compute_phf<N, M>(keys, result)) { return true; }
+    constexpr std::size_t NextM = M * 2;
+    if constexpr (NextM <= MaxM) { return try_gperf_po2<N, NextM, MaxM>(keys, result); }
+    return false;
+}
+
+// --- Hash-and-Displace fallback --------------------------------------------
+
+constexpr std::size_t hd_bucket_hash(std::string_view key) noexcept {
+    std::size_t c0 = key.empty() ? 0 : static_cast<unsigned char>(key[0]);
+    std::size_t c1 = key.empty() ? 0 : static_cast<unsigned char>(key[key.size() - 1]);
+    return (c0 + c1 * 3 + key.size() * 17) & 0xFF;
+}
+constexpr std::size_t hd_safe_char(const char* p, std::size_t len, std::size_t idx) noexcept {
+    std::size_t has = static_cast<std::size_t>(idx < len);
+    std::size_t si = idx & (std::size_t{0} - has);
+    return static_cast<unsigned char>(p[si]) & (std::size_t{0} - has);
+}
+constexpr std::size_t hd_key_hash_2(std::string_view key) noexcept {
+    std::size_t kc = key.size();
+    kc = kc * 31 + static_cast<unsigned char>(key[0]);
+    kc = kc * 31 + hd_safe_char(key.data(), key.size(), 1);
+    return kc;
+}
+constexpr std::size_t hd_key_hash_4(std::string_view key) noexcept {
+    std::size_t kc = key.size();
+    kc = kc * 31 + static_cast<unsigned char>(key[0]);
+    kc = kc * 31 + hd_safe_char(key.data(), key.size(), 1);
+    kc = kc * 31 + hd_safe_char(key.data(), key.size(), 2);
+    kc = kc * 31 + hd_safe_char(key.data(), key.size(), 3);
+    return kc;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_hash_and_displace(
+    const std::array<std::string_view, N>& keys,
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS>& asso_values,
+    std::size_t& num_positions,
+    std::array<std::size_t, MAX_POSITIONS>& positions,
+    std::array<std::size_t, M>& slot_to_key) {
+    num_positions = select_positions<N>(keys, positions, M);
+
+    if (num_positions == 0) {
+        for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+        for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+        for (std::size_t i = 0; i < N; ++i) {
+            std::size_t slot = keys[i].size() % M;
+            if (slot_to_key[slot] != N) { return false; }
+            slot_to_key[slot] = i;
+        }
+        return true;
+    }
+
+    for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+    num_positions = HD_MODE; // sentinel for H&D mode
+    positions[0] = 0;
+    positions[1] = LAST_CHAR;
+
+    std::array<std::size_t, N> key_bucket{};
+    for (std::size_t i = 0; i < N; ++i) { key_bucket[i] = hd_bucket_hash(keys[i]); }
+
+    struct bucket_info { std::size_t ch; std::size_t count; };
+    std::array<bucket_info, N> buckets{};
+    std::size_t num_buckets = 0;
+    for (std::size_t i = 0; i < N; ++i) {
+        std::size_t bk = key_bucket[i];
+        bool found = false;
+        for (std::size_t b = 0; b < num_buckets; ++b) {
+            if (buckets[b].ch == bk) { ++buckets[b].count; found = true; break; }
+        }
+        if (!found) { buckets[num_buckets++] = {bk, 1}; }
+    }
+    for (std::size_t i = 0; i < num_buckets; ++i) {
+        for (std::size_t j = i + 1; j < num_buckets; ++j) {
+            if (buckets[j].count > buckets[i].count) {
+                auto tmp = buckets[i]; buckets[i] = buckets[j]; buckets[j] = tmp;
+            }
+        }
+    }
+
+    auto try_placement = [&](auto key_hash_fn) -> bool {
+        for (std::size_t i = 0; i < M; ++i) { slot_to_key[i] = N; }
+        for (std::size_t i = 0; i < 256; ++i) { asso_values[0][i] = 0; }
+        for (std::size_t b = 0; b < num_buckets; ++b) {
+            std::size_t ch = buckets[b].ch;
+            std::array<std::size_t, N> bucket_keys{};
+            std::size_t bk_count = 0;
+            for (std::size_t i = 0; i < N; ++i) {
+                if (key_bucket[i] == ch) { bucket_keys[bk_count++] = i; }
+            }
+            bool placed = false;
+            std::size_t max_d = M < 255 ? M : 255;
+            for (std::size_t d = 0; d < max_d; ++d) {
+                bool ok = true;
+                std::array<std::size_t, N> bucket_slots{};
+                for (std::size_t k = 0; k < bk_count; ++k) {
+                    std::size_t slot = (d + key_hash_fn(keys[bucket_keys[k]])) % M;
+                    if (slot_to_key[slot] != N) { ok = false; break; }
+                    for (std::size_t k2 = 0; k2 < k; ++k2) {
+                        if (bucket_slots[k2] == slot) { ok = false; break; }
+                    }
+                    if (!ok) { break; }
+                    bucket_slots[k] = slot;
+                }
+                if (ok) {
+                    asso_values[0][ch] = d;
+                    for (std::size_t k = 0; k < bk_count; ++k) {
+                        slot_to_key[bucket_slots[k]] = bucket_keys[k];
+                    }
+                    placed = true;
+                    break;
+                }
+            }
+            if (!placed) { return false; }
+        }
+        std::size_t filled = 0;
+        for (std::size_t i = 0; i < M; ++i) {
+            if (slot_to_key[i] != N) { ++filled; }
+        }
+        return filled == N;
+    };
+
+    if (try_placement([](std::string_view k) { return hd_key_hash_2(k); })) {
+        positions[2] = HD_HASH_2BYTE_FLAG;
+        return true;
+    }
+    if (try_placement([](std::string_view k) { return hd_key_hash_4(k); })) {
+        positions[2] = HD_HASH_4BYTE_FLAG;
+        return true;
+    }
+    return false;
+}
+
+template <std::size_t N, std::size_t M>
+consteval bool try_compute_phf_hd(const std::array<std::string_view, N>& keys, phf_result<N>& result) {
+    static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+    std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso{};
+    std::size_t npos{};
+    std::array<std::size_t, MAX_POSITIONS> pos{};
+    std::array<std::size_t, M> s2k{};
+    if (try_hash_and_displace<N, M>(keys, asso, npos, pos, s2k)) {
+        result.table_size = M;
+        result.asso_values = asso;
+        result.num_positions = npos;
+        result.positions = pos;
+        for (std::size_t i = 0; i < M; ++i) { result.slot_to_key[i] = s2k[i]; }
+        for (std::size_t i = M; i < phf_result<N>::MAX_TABLE_SIZE; ++i) { result.slot_to_key[i] = N; }
+        return true;
+    }
+    return false;
+}
+
+template <std::size_t N, std::size_t M>
+consteval phf_result<N> compute_phf_hd_po2(const std::array<std::string_view, N>& keys) {
+    static_assert(M <= phf_result<N>::MAX_TABLE_SIZE, "Table size M exceeds maximum");
+    phf_result<N> result{};
+    if (try_compute_phf_hd<N, M>(keys, result)) { return result; }
+    constexpr std::size_t NextM = M * 2;
+    if constexpr (NextM <= phf_result<N>::MAX_TABLE_SIZE) {
+        return compute_phf_hd_po2<N, NextM>(keys);
+    } else {
+        compile_time_error("Hash-and-Displace: failed to find valid table size");
+        return result;
+    }
+}
+
+// Compute a perfect hash for `keys`: try gperf at power-of-two sizes (capped so
+// the runtime tables stay within uint8 indices), then fall back to H&D.
+template <std::size_t N>
+consteval phf_result<N> compute_phf(const std::array<std::string_view, N>& keys) {
+    constexpr std::size_t StartM = next_power_of_2(N);
+    constexpr std::size_t GPERF_MAX_TABLE =
+        phf_result<N>::MAX_TABLE_SIZE < 256 ? phf_result<N>::MAX_TABLE_SIZE : 256;
+    if constexpr (StartM <= GPERF_MAX_TABLE) {
+        phf_result<N> result{};
+        if (try_gperf_po2<N, StartM, GPERF_MAX_TABLE>(keys, result)) { return result; }
+    }
+    return compute_phf_hd_po2<N, StartM>(keys);
+}
+
+// ============================================================================
+// Runtime tables (flat, uint8) derived from a phf_result.
+// ============================================================================
+
+template <std::size_t N, std::size_t TableSize, std::size_t MaxKeyLen>
+struct phf_data {
+    std::array<std::array<std::uint8_t, 256>, MAX_POSITIONS> asso_values{};
+    std::array<std::uint8_t, MAX_POSITIONS>                  positions{};
+    std::uint8_t                                             num_positions{};
+    std::uint8_t                                             hd_hash_variant{}; // 2 or 4 (H&D only)
+    std::array<std::uint8_t, TableSize>                      slot_to_key{};
+    // slot_key_bytes[s] holds the key stored at slot s, zero-padded to a 16-byte
+    // multiple so the SIMD comparison can read a whole register.
+    std::array<std::array<char, ((MaxKeyLen + 15) / 16) * 16>, TableSize> slot_key_bytes{};
+    std::array<std::uint8_t, TableSize>                      slot_key_len{};
+};
+
+template <std::size_t N>
+constexpr std::size_t compute_max_key_len(const std::array<std::string_view, N>& keys) noexcept {
+    std::size_t m = 0;
+    for (std::size_t i = 0; i < N; ++i) { if (keys[i].size() > m) { m = keys[i].size(); } }
+    return m;
+}
+
+// Validate keys and build the runtime tables from the computed perfect hash.
+template <std::size_t N, std::size_t TableSize, std::size_t MaxKeyLen>
+consteval phf_data<N, TableSize, MaxKeyLen>
+build_phf_data(const std::array<std::string_view, N>& keys, const phf_result<N>& result) {
+    for (std::size_t i = 0; i < N; ++i) {
+        if (keys[i].empty())            { compile_time_error("empty keys are not allowed in key_selector"); }
+        if (keys[i].size() > MaxKeyLen) { compile_time_error("key length exceeds MaxKeyLen"); }
+        for (char c : keys[i]) {
+            if (c == '\\') { compile_time_error("backslash not allowed in key_selector keys"); }
+            if (c == '"')  { compile_time_error("quote not allowed in key_selector keys"); }
+            if (c == '\0') { compile_time_error("null byte not allowed in key_selector keys"); }
+        }
+        for (std::size_t j = i + 1; j < N; ++j) {
+            if (keys[i] == keys[j]) { compile_time_error("duplicate keys in key_selector"); }
+        }
+    }
+
+    phf_data<N, TableSize, MaxKeyLen> out{};
+
+    if (result.num_positions == HD_MODE) {
+        // H&D mode: single displacement table in asso_values[0].
+        for (std::size_t c = 0; c < 256; ++c) {
+            out.asso_values[0][c] = static_cast<std::uint8_t>(result.asso_values[0][c]);
+        }
+        out.num_positions   = static_cast<std::uint8_t>(HD_MODE);
+        out.hd_hash_variant = static_cast<std::uint8_t>(result.positions[2]);
+    } else {
+        for (std::size_t pi = 0; pi < result.num_positions; ++pi) {
+            for (std::size_t c = 0; c < 256; ++c) {
+                out.asso_values[pi][c] = static_cast<std::uint8_t>(result.asso_values[pi][c] % TableSize);
+            }
+        }
+        out.num_positions = static_cast<std::uint8_t>(result.num_positions);
+        for (std::size_t i = 0; i < result.num_positions; ++i) {
+            out.positions[i] = (result.positions[i] == LAST_CHAR)
+                ? POS_LAST_CHAR
+                : static_cast<std::uint8_t>(result.positions[i]);
+        }
+    }
+
+    for (std::size_t s = 0; s < TableSize; ++s) {
+        out.slot_to_key[s] = static_cast<std::uint8_t>(result.slot_to_key[s]);
+    }
+
+    for (std::size_t s = 0; s < TableSize; ++s) {
+        std::size_t ki = result.slot_to_key[s];
+        if (ki < N) {
+            auto k = keys[ki];
+            out.slot_key_len[s] = static_cast<std::uint8_t>(k.size());
+            for (std::size_t c = 0; c < k.size(); ++c) { out.slot_key_bytes[s][c] = k[c]; }
+        } else {
+            out.slot_key_len[s] = 0; // empty slot: no length can match
+        }
+    }
+    return out;
+}
+
+// --- SIMD runtime primitives ------------------------------------------------
+
+// The runtime matchers read whole 16-byte blocks up to offset 63 (four blocks)
+// for keys as long as the 63-character maximum. That read must stay within the
+// buffer's trailing padding, so the padding has to exceed the largest offset we
+// touch.
+static_assert(SIMDJSON_PADDING > 63,
+              "key_selector requires SIMDJSON_PADDING > 63 for its SIMD key reads");
+
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+// True when every byte of `diff` is zero. A 32-bit-lane horizontal max is
+// enough (zero iff every 32-bit word is zero) and is cheaper than a byte-wide
+// reduction. Do not replace this with a floating-point compare against 0.0:
+// flush-to-zero would treat a denormal as zero.
+simdjson_really_inline bool neon_all_bytes_zero(uint8x16_t diff) noexcept {
+    return vmaxvq_u32(vreinterpretq_u32_u8(diff)) == 0;
+}
+#endif
+
+// Scan for the terminating '"' starting at p. Returns its byte offset (= key
+// length). Caller guarantees SIMDJSON_PADDING (== 64) bytes past the JSON buffer,
+// so reading whole 16-byte blocks up to offset 63 is always safe.
+template <std::size_t MaxKeyLen>
+simdjson_really_inline std::size_t scan_key_length(const char* p) noexcept {
+    // The closing quote of the longest key sits at most at offset MaxKeyLen, so we
+    // scan as many 16-byte blocks as it takes to cover offsets 0..MaxKeyLen. The
+    // cap (63) keeps every covered offset within the 64-byte padding guarantee, so
+    // the SIMD and scalar builds agree.
+    static_assert(MaxKeyLen <= 63, "MaxKeyLen must be <= 63 for current SIMD implementations");
+    // [[maybe_unused]]: the scalar fallback (no NEON/SSE2/LSX) does not use it.
+    [[maybe_unused]] constexpr std::size_t num_blocks = MaxKeyLen / 16 + 1;
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+    for (std::size_t b = 0; b < num_blocks; ++b) {
+        uint8x16_t v = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + b * 16);
+        uint8x16_t cmp = vceqq_u8(v, vdupq_n_u8('"'));
+        uint64_t m = vget_lane_u64(
+            vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(cmp), 4)), 0);
+        if (simdjson_likely(m != 0)) { return b * 16 + (std::size_t(__builtin_ctzll(m)) >> 2); }
+    }
+    return MaxKeyLen + 1;
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+    for (std::size_t b = 0; b < num_blocks; ++b) {
+        __m128i v   = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + b * 16));
+        __m128i cmp = _mm_cmpeq_epi8(v, _mm_set1_epi8('"'));
+        unsigned m  = static_cast<unsigned>(_mm_movemask_epi8(cmp));
+        if (simdjson_likely(m != 0)) { return b * 16 + std::size_t(__builtin_ctz(m)); }
+    }
+    return MaxKeyLen + 1;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+    for (std::size_t b = 0; b < num_blocks; ++b) {
+        __m128i v   = __lsx_vld(reinterpret_cast<const void*>(p + b * 16), 0);
+        __m128i cmp = __lsx_vseq_b(v, __lsx_vreplgr2vr_b('"'));
+        // vmskltz_b gathers the per-byte sign bits (set where the byte equals '"')
+        // into the low 16 bits of lane 0, the LSX equivalent of movemask.
+        unsigned m  = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(cmp), 0)) & 0xFFFFu;
+        if (simdjson_likely(m != 0)) { return b * 16 + std::size_t(__builtin_ctz(m)); }
+    }
+    return MaxKeyLen + 1;
+#else
+    for (std::size_t i = 0; i <= MaxKeyLen; ++i)
+        if (p[i] == '"') return i;
+    return MaxKeyLen + 1;
+#endif
+}
+
+// Byte-equal of p[0..len) against stored[0..len). stored is zero-padded past
+// `len`. Input is read over 16, 32, 48, or 64 bytes (padded JSON buffer
+// guaranteed; SIMDJSON_PADDING == 64).
+template <std::size_t MaxKeyLen>
+simdjson_really_inline bool compare_key_bytes(
+    const char* p, const char* stored, std::size_t len) noexcept {
+    // [[maybe_unused]]: only the NEON/SSE2/LSX branches read these; the scalar
+    // fallback build (no SIMD) leaves them unused, which is an error under -Werror.
+    [[maybe_unused]] alignas(16) static constexpr uint8_t idx16[16] =
+        {0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15};
+    if constexpr (MaxKeyLen <= 16) {
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+        uint8x16_t vp = vld1q_u8(reinterpret_cast<const uint8_t*>(p));
+        uint8x16_t vs = vld1q_u8(reinterpret_cast<const uint8_t*>(stored));
+        uint8x16_t mask = vcltq_u8(vld1q_u8(idx16), vdupq_n_u8(static_cast<uint8_t>(len)));
+        uint8x16_t diff = veorq_u8(vandq_u8(vp, mask), vs);
+        return neon_all_bytes_zero(diff);
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+        __m128i vp = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p));
+        __m128i vs = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored));
+        __m128i idx = _mm_load_si128(reinterpret_cast<const __m128i*>(idx16));
+        __m128i mask = _mm_cmplt_epi8(idx, _mm_set1_epi8(static_cast<char>(len)));
+        __m128i eq = _mm_cmpeq_epi8(_mm_and_si128(vp, mask), vs);
+        return _mm_movemask_epi8(eq) == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+        __m128i vp = __lsx_vld(reinterpret_cast<const void*>(p), 0);
+        __m128i vs = __lsx_vld(reinterpret_cast<const void*>(stored), 0);
+        __m128i idx = __lsx_vld(reinterpret_cast<const void*>(idx16), 0);
+        __m128i mask = __lsx_vslt_b(idx, __lsx_vreplgr2vr_b(static_cast<int>(len)));
+        __m128i eq = __lsx_vseq_b(__lsx_vand_v(vp, mask), vs);
+        return (static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq), 0)) & 0xFFFFu) == 0xFFFFu;
+#else
+        for (std::size_t i = 0; i < len; ++i)
+            if (p[i] != stored[i]) return false;
+        return true;
+#endif
+    } else if constexpr (MaxKeyLen <= 32) {
+        [[maybe_unused]] alignas(16) static constexpr uint8_t idx32_hi[16] =
+            {16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31};
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+        uint8x16_t vp_lo = vld1q_u8(reinterpret_cast<const uint8_t*>(p));
+        uint8x16_t vp_hi = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + 16);
+        uint8x16_t vs_lo = vld1q_u8(reinterpret_cast<const uint8_t*>(stored));
+        uint8x16_t vs_hi = vld1q_u8(reinterpret_cast<const uint8_t*>(stored) + 16);
+        uint8x16_t lenv = vdupq_n_u8(static_cast<uint8_t>(len));
+        uint8x16_t m_lo = vcltq_u8(vld1q_u8(idx16),    lenv);
+        uint8x16_t m_hi = vcltq_u8(vld1q_u8(idx32_hi), lenv);
+        uint8x16_t d_lo = veorq_u8(vandq_u8(vp_lo, m_lo), vs_lo);
+        uint8x16_t d_hi = veorq_u8(vandq_u8(vp_hi, m_hi), vs_hi);
+        return neon_all_bytes_zero(vorrq_u8(d_lo, d_hi));
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+        __m128i vp_lo = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p));
+        __m128i vp_hi = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + 16));
+        __m128i vs_lo = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored));
+        __m128i vs_hi = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored + 16));
+        __m128i lenv = _mm_set1_epi8(static_cast<char>(len));
+        __m128i m_lo = _mm_cmplt_epi8(_mm_load_si128(reinterpret_cast<const __m128i*>(idx16)),    lenv);
+        __m128i m_hi = _mm_cmplt_epi8(_mm_load_si128(reinterpret_cast<const __m128i*>(idx32_hi)), lenv);
+        __m128i eq_lo = _mm_cmpeq_epi8(_mm_and_si128(vp_lo, m_lo), vs_lo);
+        __m128i eq_hi = _mm_cmpeq_epi8(_mm_and_si128(vp_hi, m_hi), vs_hi);
+        return (_mm_movemask_epi8(eq_lo) & _mm_movemask_epi8(eq_hi)) == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+        __m128i lenv = __lsx_vreplgr2vr_b(static_cast<int>(len));
+        __m128i vp_lo = __lsx_vld(reinterpret_cast<const void*>(p), 0);
+        __m128i vp_hi = __lsx_vld(reinterpret_cast<const void*>(p + 16), 0);
+        __m128i vs_lo = __lsx_vld(reinterpret_cast<const void*>(stored), 0);
+        __m128i vs_hi = __lsx_vld(reinterpret_cast<const void*>(stored + 16), 0);
+        __m128i m_lo = __lsx_vslt_b(__lsx_vld(reinterpret_cast<const void*>(idx16), 0),    lenv);
+        __m128i m_hi = __lsx_vslt_b(__lsx_vld(reinterpret_cast<const void*>(idx32_hi), 0), lenv);
+        __m128i eq_lo = __lsx_vseq_b(__lsx_vand_v(vp_lo, m_lo), vs_lo);
+        __m128i eq_hi = __lsx_vseq_b(__lsx_vand_v(vp_hi, m_hi), vs_hi);
+        unsigned mlo = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq_lo), 0)) & 0xFFFFu;
+        unsigned mhi = static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq_hi), 0)) & 0xFFFFu;
+        return (mlo & mhi) == 0xFFFFu;
+#else
+        for (std::size_t i = 0; i < len; ++i)
+            if (p[i] != stored[i]) return false;
+        return true;
+#endif
+    } else if constexpr (MaxKeyLen <= 64) {
+        // 3 or 4 16-byte blocks (33..48 -> 3, 49..64 -> 4). stored is zero-padded
+        // to exactly num_blocks*16 bytes (KEY_STRIDE), so neither load overruns it.
+        // [[maybe_unused]]: the scalar fallback (no NEON/SSE2/LSX) does not use it.
+        [[maybe_unused]] constexpr std::size_t num_blocks = (MaxKeyLen + 15) / 16;
+#if SIMDJSON_KEY_SELECTOR_HAS_NEON
+        uint8x16_t base = vld1q_u8(idx16);
+        uint8x16_t lenv = vdupq_n_u8(static_cast<uint8_t>(len));
+        uint8x16_t acc  = vdupq_n_u8(0);
+        for (std::size_t b = 0; b < num_blocks; ++b) {
+            uint8x16_t vp   = vld1q_u8(reinterpret_cast<const uint8_t*>(p) + b * 16);
+            uint8x16_t vs   = vld1q_u8(reinterpret_cast<const uint8_t*>(stored) + b * 16);
+            uint8x16_t idxv = vaddq_u8(base, vdupq_n_u8(static_cast<uint8_t>(b * 16)));
+            uint8x16_t mask = vcltq_u8(idxv, lenv);
+            acc = vorrq_u8(acc, veorq_u8(vandq_u8(vp, mask), vs));
+        }
+        return neon_all_bytes_zero(acc);
+#elif SIMDJSON_KEY_SELECTOR_HAS_SSE2
+        __m128i base = _mm_load_si128(reinterpret_cast<const __m128i*>(idx16));
+        __m128i lenv = _mm_set1_epi8(static_cast<char>(len));
+        int eq = 0xFFFF;
+        for (std::size_t b = 0; b < num_blocks; ++b) {
+            __m128i vp   = _mm_loadu_si128(reinterpret_cast<const __m128i*>(p + b * 16));
+            __m128i vs   = _mm_loadu_si128(reinterpret_cast<const __m128i*>(stored + b * 16));
+            __m128i idxv = _mm_add_epi8(base, _mm_set1_epi8(static_cast<char>(b * 16)));
+            __m128i mask = _mm_cmplt_epi8(idxv, lenv);
+            eq &= _mm_movemask_epi8(_mm_cmpeq_epi8(_mm_and_si128(vp, mask), vs));
+        }
+        return eq == 0xFFFF;
+#elif SIMDJSON_KEY_SELECTOR_HAS_LSX
+        __m128i base = __lsx_vld(reinterpret_cast<const void*>(idx16), 0);
+        __m128i lenv = __lsx_vreplgr2vr_b(static_cast<int>(len));
+        unsigned acc = 0xFFFFu;
+        for (std::size_t b = 0; b < num_blocks; ++b) {
+            __m128i vp   = __lsx_vld(reinterpret_cast<const void*>(p + b * 16), 0);
+            __m128i vs   = __lsx_vld(reinterpret_cast<const void*>(stored + b * 16), 0);
+            __m128i idxv = __lsx_vadd_b(base, __lsx_vreplgr2vr_b(static_cast<int>(b * 16)));
+            __m128i mask = __lsx_vslt_b(idxv, lenv);
+            __m128i eq   = __lsx_vseq_b(__lsx_vand_v(vp, mask), vs);
+            acc &= static_cast<unsigned>(__lsx_vpickve2gr_w(__lsx_vmskltz_b(eq), 0)) & 0xFFFFu;
+        }
+        return acc == 0xFFFFu;
+#else
+        for (std::size_t i = 0; i < len; ++i)
+            if (p[i] != stored[i]) return false;
+        return true;
+#endif
+    } else {
+        for (std::size_t i = 0; i < len; ++i)
+            if (p[i] != stored[i]) return false;
+        return true;
+    }
+}
+
+// --- Single 8-bit window fast path ------------------------------------------
+//
+// Many small key sets can be told apart by inspecting a *single* 8-bit window of
+// the key bytes -- and that window need not be byte-aligned. Because every JSON
+// key is terminated by a '"', the bytes at and before a key's length are well
+// defined for any key at least that long: byte i is the key character when i is
+// inside the key and the closing quote when i == len. So we read two bytes at a
+// fixed offset, extract 8 consecutive bits at a fixed intra-byte shift, and if
+// that value is distinct for every key, a 256-entry table maps it straight to a
+// candidate key. The match then needs no hash and -- in the length-free overload
+// -- no SIMD length scan: load two bytes, shift, mask, index the table, and
+// confirm the candidate with one comparison.
+//
+// Allowing an *unaligned* window (a shift of 1..7) mixes bits from two adjacent
+// bytes and discriminates key sets that no single aligned byte can. For example,
+// the partial_tweets keys {created_at,id,text,in_reply_to_status_id,user,
+// retweet_count,favorite_count} share a colliding byte at every aligned position
+// 0,1,2, yet the 8 bits starting at bit offset 2 are unique across all seven.
+// Simpler cases fall out as the shift==0 special case: {"id","screen_name"}
+// splits on byte 0, {"jo","joe"} on the quote at byte 2.
+//
+// The window is confined to the first (shortest key length + 1) bytes so the
+// two-byte read never crosses a key's closing quote into uncontrolled value
+// bytes; that final byte is the shortest key's quote.
+template <std::size_t N, std::size_t MaxKeyLen>
+struct window_data {
+    static constexpr std::size_t KEY_STRIDE = ((MaxKeyLen + 15) / 16) * 16;
+    bool                                            ok{false};
+    std::uint8_t                                    byte_offset{0}; // first byte of the 2-byte read
+    std::uint8_t                                    shift{0};       // intra-byte bit shift (0..7)
+    std::array<std::uint8_t, 256>                   window_to_key{}; // window byte -> key index, N if none
+    std::array<std::uint8_t, N>                     key_len{};
+    std::array<std::array<char, KEY_STRIDE>, N>     key_bytes{};     // zero-padded
+};
+
+// Byte seen at `idx` for key `i`: a key character when idx is inside the key, or
+// the closing '"' when idx == len (callers keep idx <= every key's length).
+template <std::size_t N>
+constexpr unsigned window_byte_at(const std::array<std::string_view, N>& keys,
+                                  std::size_t i, std::size_t idx) noexcept {
+    if (idx < keys[i].size()) { return static_cast<unsigned char>(keys[i][idx]); }
+    return static_cast<unsigned>('"');
+}
+
+// The 8-bit window value for key `i` at (byte_offset, shift): the little-endian
+// pair (byte[off], byte[off+1]) shifted right by `shift` and truncated. Matches
+// the runtime read exactly.
+template <std::size_t N>
+constexpr unsigned window_value(const std::array<std::string_view, N>& keys,
+                                std::size_t i, std::size_t byte_offset, std::size_t shift) noexcept {
+    unsigned lo = window_byte_at<N>(keys, i, byte_offset);
+    unsigned hi = window_byte_at<N>(keys, i, byte_offset + 1);
+    return ((lo | (hi << 8)) >> shift) & 0xFFu;
+}
+
+template <std::size_t N, std::size_t MaxKeyLen>
+consteval window_data<N, MaxKeyLen>
+compute_window(const std::array<std::string_view, N>& keys) {
+    window_data<N, MaxKeyLen> out{};
+
+    std::size_t min_len = keys[0].size();
+    for (std::size_t i = 1; i < N; ++i) {
+        if (keys[i].size() < min_len) { min_len = keys[i].size(); }
+    }
+
+    // Iterate windows nearest the front first (cheapest to read, smallest shift).
+    for (std::size_t off = 0; off <= min_len; ++off) {
+        for (std::size_t shift = 0; shift < 8; ++shift) {
+            // The read touches byte off, and byte off+1 when shift != 0. Both must
+            // stay within the safe region [0, min_len] (min_len is the shortest
+            // key's quote index). off <= min_len is guaranteed by the loop bound.
+            if (shift != 0 && off + 1 > min_len) { continue; }
+
+            bool distinct = true;
+            for (std::size_t i = 0; i < N && distinct; ++i) {
+                for (std::size_t j = i + 1; j < N; ++j) {
+                    if (window_value<N>(keys, i, off, shift) == window_value<N>(keys, j, off, shift)) {
+                        distinct = false;
+                        break;
+                    }
+                }
+            }
+            if (!distinct) { continue; }
+
+            out.ok          = true;
+            out.byte_offset = static_cast<std::uint8_t>(off);
+            out.shift       = static_cast<std::uint8_t>(shift);
+            for (std::size_t b = 0; b < 256; ++b) { out.window_to_key[b] = static_cast<std::uint8_t>(N); }
+            for (std::size_t i = 0; i < N; ++i) {
+                out.window_to_key[window_value<N>(keys, i, off, shift)] = static_cast<std::uint8_t>(i);
+                out.key_len[i] = static_cast<std::uint8_t>(keys[i].size());
+                for (std::size_t c = 0; c < keys[i].size(); ++c) { out.key_bytes[i][c] = keys[i][c]; }
+            }
+            return out;
+        }
+    }
+    return out; // ok == false: no single 8-bit window distinguishes the keys
+}
+
+// Read the 8-bit window at (byte_offset, shift) from a padded key pointer.
+simdjson_really_inline std::uint8_t read_window(const char* p, std::size_t byte_offset,
+                                                std::size_t shift) noexcept {
+    std::uint16_t w;
+    // Two controlled bytes (within the shortest key + its quote, hence within the
+    // padded buffer). memcpy is the portable little-endian unaligned load.
+    std::memcpy(&w, p + byte_offset, sizeof(w));
+#if defined(__BYTE_ORDER__) && (__BYTE_ORDER__ == __ORDER_BIG_ENDIAN__)
+    w = static_cast<std::uint16_t>((w >> 8) | (w << 8));
+#endif
+    return static_cast<std::uint8_t>((w >> shift) & 0xFFu);
+}
+
+// Verify a window candidate. The window table already mapped the key to the
+// single possible index `ki` (< N); here we confirm it. Folding over 0..N-1
+// turns the runtime `ki` into a compile-time index in the matching arm, so the
+// candidate's length and bytes are constants for compare_key_bytes -- the same
+// specialization the ordered find_field path gets from a CT-length
+// unsafe_is_equal, and what keeps this path competitive. The closing-quote check
+// (p[L] == '"') both confirms the key ends exactly at the candidate's length and
+// rejects a wrong-length key, so this works whether or not the caller knew len.
+template <std::size_t N, std::size_t MaxKeyLen, std::size_t... Is>
+simdjson_really_inline std::size_t
+match_window_candidate(const char* p, std::uint8_t ki,
+                       const window_data<N, MaxKeyLen>& w,
+                       std::index_sequence<Is...>) noexcept {
+  std::size_t result = N;
+  auto try_match = [&](auto Ic) {
+    constexpr std::size_t i = decltype(Ic)::value;
+    if (ki == i && p[w.key_len[i]] == '"' &&
+        key_selector_detail::compare_key_bytes<MaxKeyLen>(
+            p, w.key_bytes[i].data(), w.key_len[i])) {
+      result = i;
+    }
+  };
+  (try_match(std::integral_constant<std::size_t, Is>{}), ...);
+  return result;
+}
+
+// --- describe() string helpers (constexpr; no std::to_string, which is not) ---
+
+// Append the decimal form of v to s.
+SIMDJSON_CONSTEXPR_STRING void append_uint(std::string& s, std::size_t v) {
+    if (v == 0) { s.push_back('0'); return; }
+    char buf[20];
+    std::size_t n = 0;
+    while (v > 0) { buf[n++] = static_cast<char>('0' + (v % 10)); v /= 10; }
+    while (n > 0) { s.push_back(buf[--n]); }
+}
+
+// Append a byte as its decimal value, plus the printable character in quotes
+// when it is in the printable ASCII range (e.g. "110 ('n')").
+SIMDJSON_CONSTEXPR_STRING void append_byte(std::string& s, unsigned b) {
+    append_uint(s, b);
+    if (b >= 0x20 && b < 0x7f) {
+        s += " ('";
+        s.push_back(static_cast<char>(b));
+        s += "')";
+    }
+}
+
+} // namespace key_selector_detail
+
+/**
+ * Stateless, compile-time key selector.
+ *
+ * Usage:
+ *   using sel_t = key_selector<"id", "text", "user">;
+ *   std::size_t i = sel_t::match_raw(raw_key); // returns sel_t::size() on miss
+ *
+ * The perfect hash is built at compile time (gperf-style, with a
+ * Hash-and-Displace fallback) and only flat tables survive to runtime. All
+ * tables are static constexpr, so the lookup fully inlines.
+ *
+ * Limitations:
+ *   - Each key must be at most 63 characters long (and no longer than
+ *     SIMDJSON_PADDING). Longer keys trigger a compile-time error.
+ *   - The number of keys should be moderate. The hard limit is 255 keys;
+ *     compilation time grows with the number of keys, so prefer a few dozen at
+ *     most per selector.
+ *   - Keys must be distinct, non-empty, and free of backslash, double-quote and
+ *     null bytes (matching is done against the raw, unescaped JSON key bytes).
+ */
+template <constevalutil::fixed_string... Keys>
+struct key_selector {
+    static constexpr std::size_t N = sizeof...(Keys);
+    static_assert(N > 0,   "key_selector requires at least one key");
+    static_assert(N <= 255,"key_selector supports at most 255 keys");
+
+    static constexpr std::array<std::string_view, N> keys{ Keys.view()... };
+    static constexpr std::size_t max_key_len = key_selector_detail::compute_max_key_len<N>(keys);
+    static_assert(max_key_len <= SIMDJSON_PADDING,
+                  "key longer than SIMDJSON_PADDING is not supported");
+    // The SIMD key-length scan covers offsets 0..63 (four 16-byte blocks), which
+    // stays within the 64-byte padding guarantee. A 64-character key's closing
+    // quote would land at offset 64 and be missed on NEON/SSE2/LSX while still
+    // matching in scalar builds, so cap at 63 to keep implementations in agreement.
+    static_assert(max_key_len <= 63,
+                  "key_selector keys must be at most 63 characters long");
+
+    static constexpr auto result = key_selector_detail::compute_phf<N>(keys);
+    static constexpr std::size_t table_size = result.table_size;
+
+    static constexpr auto phf =
+        key_selector_detail::build_phf_data<N, table_size, max_key_len>(keys, result);
+
+    // Single 8-bit-window discriminator (when one exists). Detected at compile
+    // time and selected with `if constexpr` below, so the hash path is compiled
+    // out for key sets that qualify, and this is compiled out for those that do
+    // not.
+    static constexpr auto window =
+        key_selector_detail::compute_window<N, max_key_len>(keys);
+
+    static constexpr std::size_t size() noexcept { return N; }
+
+    /**
+     * Look up a JSON key whose length is already known. p must point at the first
+     * key byte (just after the opening quote) in a padded simdjson buffer, and len
+     * must be the number of raw key bytes (the distance to the closing quote).
+     * Returns the selector index in [0, N) on match, or N on miss.
+     *
+     * Prefer this overload when the caller can obtain the key length cheaply (for
+     * example, object::for_each derives it from the structural index rather than
+     * re-scanning for the closing quote).
+     */
+    static simdjson_really_inline std::size_t match_raw(const char* p, std::size_t len) noexcept {
+        if (len == 0 || len > max_key_len) { return N; }
+
+        if constexpr (window.ok) {
+            // One 8-bit window selects the only possible candidate key;
+            // match_window_candidate confirms it (bytes + closing quote). p sits
+            // in a padded buffer and the window stays within the shortest key +
+            // quote, so the two-byte read is always in bounds. len is unused here
+            // because the quote check already pins the key's end.
+            std::uint8_t ki = window.window_to_key[
+                key_selector_detail::read_window(p, window.byte_offset, window.shift)];
+            if (ki >= N) { return N; }
+            return key_selector_detail::match_window_candidate(
+                p, ki, window, std::make_index_sequence<N>{});
+        }
+
+        std::size_t slot;
+        if (phf.num_positions == key_selector_detail::HD_MODE) {
+            // Hash-and-Displace: bucket displacement + per-key hash.
+            std::string_view key(p, len);
+            std::size_t bucket = key_selector_detail::hd_bucket_hash(key);
+            std::size_t kh = (phf.hd_hash_variant == key_selector_detail::HD_HASH_2BYTE_FLAG)
+                ? key_selector_detail::hd_key_hash_2(key)
+                : key_selector_detail::hd_key_hash_4(key);
+            slot = (phf.asso_values[0][bucket] + kh) & (table_size - 1);
+        } else {
+            // gperf: h = len + sum of asso_values over the selected positions.
+            // positions / num_positions / asso_values are compile-time constants,
+            // so this loop fully unrolls. The idx < len guard mirrors the
+            // generator's char_at()-> 256 -> skip behavior for out-of-range
+            // positions (required: arbitrary positions may exceed a key's length).
+            std::size_t h = len;
+            for (std::uint8_t i = 0; i < phf.num_positions; ++i) {
+                std::uint8_t pos = phf.positions[i];
+                std::size_t idx = (pos == key_selector_detail::POS_LAST_CHAR)
+                                  ? (len - std::size_t{1})
+                                  : static_cast<std::size_t>(pos);
+                if (idx < len) {
+                    h += phf.asso_values[i][static_cast<unsigned char>(p[idx])];
+                }
+            }
+            slot = h & (table_size - 1);
+        }
+
+        std::uint8_t ki = phf.slot_to_key[slot];
+        if (ki >= N) { return N; }
+        if (phf.slot_key_len[slot] != len) { return N; }
+        if (!key_selector_detail::compare_key_bytes<max_key_len>(
+                p, phf.slot_key_bytes[slot].data(), len)) { return N; }
+        return ki;
+    }
+
+    /**
+     * Look up a JSON key. rjs must point just after an opening quote in a padded
+     * simdjson buffer. Returns the selector index in [0, N) on match, or N on miss.
+     * The key length is recovered with a SIMD scan for the closing quote; callers
+     * that already know the length should use the (p, len) overload above.
+     */
+    static simdjson_really_inline std::size_t match_raw(raw_json_string rjs) noexcept {
+        const char* p = rjs.raw();
+        if constexpr (window.ok) {
+            // One 8-bit window picks the candidate; verifying the candidate's
+            // bytes and its closing '"' confirms the full key, so the length scan
+            // is unnecessary. The window read is in bounds (padding), and the
+            // candidate length is at most max_key_len.
+            std::uint8_t ki = window.window_to_key[
+                key_selector_detail::read_window(p, window.byte_offset, window.shift)];
+            if (ki >= N) { return N; }
+            return key_selector_detail::match_window_candidate(
+                p, ki, window, std::make_index_sequence<N>{});
+        }
+        return match_raw(p, key_selector_detail::scan_key_length<max_key_len>(p));
+    }
+
+    /** Return the key text at selector index i (i in [0, N)). */
+    static constexpr std::string_view key_at(std::size_t i) noexcept {
+        return keys[i];
+    }
+
+    /**
+     * Return a complete, human-readable, multi-line description of how this
+     * selector classifies a key: which algorithm was selected at compile time
+     * (single 8-bit window, gperf-style perfect hash, or hash-and-displace), the
+     * exact bytes/positions it inspects, and the contents of the lookup tables
+     * (which window bytes or hash slots map to which key). The text mirrors what
+     * match_raw() does step by step.
+     *
+     * Everything it reports is derived from the compile-time tables, so describe()
+     * is itself usable in a constant expression when the standard library supports
+     * constexpr std::string (__cpp_lib_constexpr_string):
+     *
+     *   static_assert(!key_selector<"name", "city">::describe().empty());
+     *
+     * It allocates a std::string and is meant for documentation, debugging and
+     * tests, not for any hot path.
+     */
+    static SIMDJSON_CONSTEXPR_STRING std::string describe() {
+        std::string s;
+        s += "key_selector: ";
+        key_selector_detail::append_uint(s, N);
+        s += " keys, max key length ";
+        key_selector_detail::append_uint(s, max_key_len);
+        s += "\nkeys:\n";
+        for (std::size_t i = 0; i < N; ++i) {
+            s += "  [";
+            key_selector_detail::append_uint(s, i);
+            s += "] \"";
+            s += keys[i];
+            s += "\" (length ";
+            key_selector_detail::append_uint(s, keys[i].size());
+            s += ")\n";
+        }
+        if constexpr (window.ok) {
+            // Mirrors the window fast path of match_raw().
+            s += "algorithm: single 8-bit window\n";
+            s += "  step 1: read 2 bytes at offset ";
+            key_selector_detail::append_uint(s, window.byte_offset);
+            s += ", interpret them as a little-endian 16-bit value, shift right by ";
+            key_selector_detail::append_uint(s, window.shift);
+            s += " bits, and keep the low 8 bits\n";
+            s += "  step 2: map that byte through a 256-entry table to a key index (";
+            key_selector_detail::append_uint(s, N);
+            s += " means no match):\n";
+            for (std::size_t b = 0; b < 256; ++b) {
+                if (window.window_to_key[b] < N) {
+                    s += "    byte ";
+                    key_selector_detail::append_byte(s, static_cast<unsigned>(b));
+                    s += " -> key ";
+                    key_selector_detail::append_uint(s, window.window_to_key[b]);
+                    s += "\n";
+                }
+            }
+            s += "  step 3: confirm the candidate by checking the closing quote sits at the key's length and comparing the key bytes\n";
+        } else {
+            // Mirrors the perfect-hash path of match_raw().
+            if constexpr (phf.num_positions == key_selector_detail::HD_MODE) {
+                s += "algorithm: hash-and-displace perfect hash\n";
+                s += "  step 1: bucket = (first_byte + last_byte*3 + length*17) mod 256\n";
+                s += "  step 2: keyhash = base-31 rolling hash of the length and the first ";
+                key_selector_detail::append_uint(s, phf.hd_hash_variant);
+                s += " bytes\n";
+                s += "  step 3: slot = (displacement[bucket] + keyhash) mod ";
+                key_selector_detail::append_uint(s, table_size);
+                s += "\n  step 4: slot_to_key[slot] gives the candidate key index\n";
+                s += "  per-key derivation:\n";
+                for (std::size_t i = 0; i < N; ++i) {
+                    std::string_view k = keys[i];
+                    std::size_t bucket = key_selector_detail::hd_bucket_hash(k);
+                    std::size_t kh = (phf.hd_hash_variant == key_selector_detail::HD_HASH_2BYTE_FLAG)
+                        ? key_selector_detail::hd_key_hash_2(k)
+                        : key_selector_detail::hd_key_hash_4(k);
+                    std::size_t slot = (phf.asso_values[0][bucket] + kh) & (table_size - 1);
+                    s += "    \"";
+                    s += k;
+                    s += "\": bucket=";
+                    key_selector_detail::append_uint(s, bucket);
+                    s += " displacement=";
+                    key_selector_detail::append_uint(s, phf.asso_values[0][bucket]);
+                    s += " keyhash=";
+                    key_selector_detail::append_uint(s, kh);
+                    s += " slot=";
+                    key_selector_detail::append_uint(s, slot);
+                    s += "\n";
+                }
+            } else {
+                s += "algorithm: gperf-style perfect hash over ";
+                key_selector_detail::append_uint(s, phf.num_positions);
+                s += " character position(s)\n";
+                s += "  step 1: h = key length\n";
+                s += "  step 2: for each position below, add its association value for the key byte there (a position past the key's length contributes 0):\n";
+                for (std::size_t i = 0; i < phf.num_positions; ++i) {
+                    s += "    position ";
+                    if (phf.positions[i] == key_selector_detail::POS_LAST_CHAR) {
+                        s += "last character";
+                    } else {
+                        s += "byte index ";
+                        key_selector_detail::append_uint(s, phf.positions[i]);
+                    }
+                    s += "\n";
+                }
+                s += "  step 3: slot = h mod ";
+                key_selector_detail::append_uint(s, table_size);
+                s += " (a power of two, applied as a bitmask)\n";
+                s += "  step 4: slot_to_key[slot] gives the candidate key index\n";
+                s += "  per-key derivation:\n";
+                for (std::size_t i = 0; i < N; ++i) {
+                    std::string_view k = keys[i];
+                    std::size_t h = k.size();
+                    for (std::size_t pi = 0; pi < phf.num_positions; ++pi) {
+                        std::size_t pos = phf.positions[pi];
+                        std::size_t idx = (pos == key_selector_detail::POS_LAST_CHAR)
+                                          ? (k.size() - 1) : pos;
+                        if (idx < k.size()) {
+                            h += phf.asso_values[pi][static_cast<unsigned char>(k[idx])];
+                        }
+                    }
+                    std::size_t slot = h & (table_size - 1);
+                    s += "    \"";
+                    s += k;
+                    s += "\": h=";
+                    key_selector_detail::append_uint(s, h);
+                    s += " slot=";
+                    key_selector_detail::append_uint(s, slot);
+                    s += "\n";
+                }
+            }
+            s += "  occupied slots (slot -> key):\n";
+            for (std::size_t slot = 0; slot < table_size; ++slot) {
+                if (phf.slot_to_key[slot] < N) {
+                    s += "    slot ";
+                    key_selector_detail::append_uint(s, slot);
+                    s += " -> key ";
+                    key_selector_detail::append_uint(s, phf.slot_to_key[slot]);
+                    s += " (\"";
+                    s += keys[phf.slot_to_key[slot]];
+                    s += "\", length ";
+                    key_selector_detail::append_uint(s, phf.slot_key_len[slot]);
+                    s += ")\n";
+                }
+            }
+            s += "  confirm the candidate by checking the key length matches and comparing the key bytes\n";
+        }
+        return s;
+    }
+};
+
+namespace key_selector_detail {
+template <typename> struct is_key_selector : std::false_type {};
+template <constevalutil::fixed_string... Keys>
+struct is_key_selector<key_selector<Keys...>> : std::true_type {};
+} // namespace key_selector_detail
+
+/**
+ * Matches any instantiation of key_selector<Keys...>. Used to constrain
+ * object::for_each so that passing a non-selector type yields a clear
+ * constraint error rather than a cascade of failures inside for_each.
+ */
+template <typename T>
+concept key_selector_type = key_selector_detail::is_key_selector<T>::value;
+
+} // namespace ondemand
+} // namespace rvv_vls
+} // namespace simdjson
+
+#endif // SIMDJSON_SUPPORTS_CONCEPTS
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
+/* end file simdjson/generic/ondemand/key_selector.h for rvv_vls */
 /* including simdjson/generic/ondemand/object.h for rvv_vls: #include "simdjson/generic/ondemand/object.h" */
 /* begin file simdjson/generic/ondemand/object.h for rvv_vls */
 #ifndef SIMDJSON_GENERIC_ONDEMAND_OBJECT_H
@@ -178238,6 +228681,7 @@ public:
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/implementation_simdjson_result_base.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/key_selector.h" */
 /* amalgamation skipped (editor-only): #include <vector> */
 /* amalgamation skipped (editor-only): #if SIMDJSON_STATIC_REFLECTION && SIMDJSON_SUPPORTS_CONCEPTS */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_string_builder.h"  // for constevalutil::fixed_string */
@@ -178248,6 +228692,114 @@ namespace simdjson {
 namespace rvv_vls {
 namespace ondemand {

+#if SIMDJSON_SUPPORTS_CONCEPTS
+/**
+ * Result of object::for_each: the first error encountered (SUCCESS if none) and
+ * the number of distinct selector keys that matched during the walk. A
+ * matched_count equal to Selector::size() means every selected key was present
+ * in the object. Implicitly converts to error_code so existing callers that only
+ * care about the error (including SIMDJSON_TRY and the test ASSERT_* macros) keep
+ * working unchanged.
+ *
+ * On error paths (a handler returns a non-SUCCESS error_code, or a direct-target
+ * deserialization via value::get fails), matched_count reflects only the number
+ * of distinct selected keys that were successfully handled *before* the failure;
+ * the failing key is not counted. This is the current behavior.
+ *
+ * Marked [[nodiscard]]: for_each surfaces parse/type errors (from a matched
+ * value, a returning handler, or the structural walk itself) only through this
+ * result, so it must not be silently dropped. This matters most in builds
+ * without exceptions, where it is the only channel for those errors. To
+ * deliberately ignore it, assign to a variable or cast to void.
+ */
+struct [[nodiscard]] for_each_result {
+  error_code error{SUCCESS};
+  std::size_t matched_count{0};
+  constexpr operator error_code() const noexcept { return error; }
+};
+
+namespace key_selector_for_each_detail {
+/**
+ * A per-key handler for the variadic object::for_each is either:
+ *   - an invocable taking a value (run custom logic for that field), or
+ *   - a deserialization target T, in which case the matched value is assigned
+ *     directly via value::get(T&).
+ * The target form lets callers bind fields straight to variables without a
+ * lambda per key -- e.g. obj.for_each<"name","city","age">(name, city, age) --
+ * while still allowing a lambda in any position when custom logic is needed
+ * (for example to descend into a nested object).
+ */
+template <typename H>
+concept field_handler =
+    std::is_invocable_v<std::remove_reference_t<H>&, value> ||
+    ::simdjson::deserializable<std::remove_cvref_t<H>, value>;
+
+/**
+ * noexcept-ness of handling a single handler: nothrow-invocability for the
+ * callback form, or nothrow-deserializability for the direct-target form
+ * (builtin scalar/string targets are noexcept; custom targets follow their
+ * own tag_invoke noexcept specification).
+ */
+template <typename H>
+inline constexpr bool nothrow_field_handler_v =
+    std::is_invocable_v<std::remove_reference_t<H>&, value>
+        ? std::is_nothrow_invocable_v<std::remove_reference_t<H>&, value>
+        : ::simdjson::nothrow_deserializable<std::remove_cvref_t<H>, value>;
+
+/**
+ * nothrow_field_handler_v folded over a whole handler pack. The variadic
+ * for_each overloads spell their conditional-noexcept specifier with this
+ * single id-expression rather than an inline fold expression. MSVC (VS18)
+ * mishandles a fold-expression that appears directly in the noexcept-specifier
+ * of an out-of-line template definition: it fails to recognize the definition's
+ * specifier as matching the in-class declaration's and reports a spurious
+ * C2382 ("redefinition; different exception specifications"). Naming the fold
+ * here keeps the specifier a plain identifier, which matches reliably.
+ */
+template <typename... Handlers>
+inline constexpr bool nothrow_field_handlers_v =
+    (nothrow_field_handler_v<Handlers> && ...);
+} // namespace key_selector_for_each_detail
+#endif
+
+/**
+ * An opaque snapshot of an object's scanning position, obtained from
+ * object::get_current_position() and consumed by object::revert_position().
+ *
+ * This bundles the raw token position with the iteration depth that was
+ * live at the moment of capture: a scalar field value (e.g. a number)
+ * unwinds back to the object's own depth once fully consumed, while a
+ * compound value (array/object) not yet fully consumed stays one level
+ * deeper. The correct depth to restore to therefore depends on what was
+ * live when the snapshot was taken, not on the object itself, which is
+ * why both pieces travel together here rather than being recomputed later.
+ *
+ * The members are private and only constructible by object itself: a
+ * hand-built value here would let revert_position() jump to an arbitrary,
+ * unvalidated position and depth, and this type carries no way to check
+ * that a given snapshot still refers to a live object. Only ever pass
+ * along a value you got from get_current_position(), and only back to the
+ * same object, before that object has been reset() or the parser has
+ * iterate()d a new document: neither invalidates a snapshot in a way this
+ * type can detect.
+ */
+struct object_position {
+  /**
+   * Default-constructed so a variable can be declared and assigned later,
+   * matching e.g. document()/object(). Not a valid position to revert to.
+   */
+  simdjson_inline object_position() noexcept = default;
+
+private:
+  token_position position{};
+  depth_t depth{};
+
+  simdjson_inline object_position(token_position position_, depth_t depth_) noexcept
+    : position(position_), depth(depth_) {}
+
+  friend class object;
+};
+
 /**
  * A forward-only JSON object field iterator.
  */
@@ -178266,8 +228818,19 @@ public:
    * Using the iterator directly is also possible but error-prone and discouraged. In particular,
    * you must dereference the iterator exactly once per iteration (before calling '++').
    * Doing otherwise is unsafe and may lead to errors. You are responsible for ensuring
+   *
+   * With SIMDJSON_DEVELOPMENT_CHECKS, the iterator locks this object while it is
+   * alive, so that reentrant access (find_field(), reset(), ...) is reported as
+   * OUT_OF_ORDER_ITERATION.
+   */
+  simdjson_inline simdjson_result<object_iterator> begin() & noexcept;
+  /**
+   * Get an iterator to the start of a temporary object, e.g., `v.get_object().begin()`.
+   *
+   * The iterator does not depend on the object instance and may outlive it, so
+   * it does not lock it.
    */
-  simdjson_inline simdjson_result<object_iterator> begin() noexcept;
+  simdjson_inline simdjson_result<object_iterator> begin() && noexcept;
   simdjson_inline simdjson_result<object_iterator> end() noexcept;
   /**
    * Look up a field by name on an object (order-sensitive). By order-sensitive, we mean that
@@ -178279,10 +228842,11 @@ public:
    *
    * ```cpp
    * simdjson::ondemand::parser parser;
-   * auto obj = parser.parse(R"( { "x": 1, "y": 2, "z": 3 } )"_padded);
-   * double z = obj.find_field("z");
-   * double y = obj.find_field("y");
-   * double x = obj.find_field("x");
+   * auto json = R"( { "x": 1, "y": 2, "z": 3 } )"_padded;
+   * auto doc = parser.iterate(json);
+   * double z = doc.find_field("z");
+   * double y = doc.find_field("y");
+   * double x = doc.find_field("x");
    * ```
    * If you have multiple fields with a matching key ({"x": 1,  "x": 1}) be mindful
    * that only one field is returned.
@@ -178355,6 +228919,100 @@ public:
   /** @overload simdjson_inline simdjson_result<value> find_field_unordered(std::string_view key) & noexcept; */
   simdjson_inline simdjson_result<value> operator[](std::string_view key) && noexcept;

+#if SIMDJSON_SUPPORTS_CONCEPTS
+  /**
+   * Walk this object once and invoke on_match(selector_index, value) for each
+   * field whose key is in the compile-time key_selector Selector, in JSON order
+   * (first occurrence of a duplicate key wins). Iteration stops once all
+   * Selector::size() keys have matched or the object ends. The value is consumed
+   * in place, so this is a low-overhead way to extract a known set of fields
+   * regardless of their order in the JSON.
+   *
+   * Like other object iteration in simdjson, for_each consumes the object by
+   * advancing the underlying iterator state; after the call the same object
+   * instance should not be used for further field access or iteration.
+   *
+   * Usage:
+   *   using sel_t = ondemand::key_selector<"id", "text", "user">;
+   *   obj.for_each<sel_t>([&](std::size_t i, ondemand::value v) {
+   *     switch (i) { case 0: ...; case 1: ...; }
+   *   });
+   *
+   * Limitations (see key_selector): each key must be at most 63 characters long,
+   * and the number of keys should be moderate (hard limit 255; a handful is
+   * best, as the compile-time perfect hash may fail or slow compilation for
+   * large key sets). The keys must be distinct, non-empty, and free of backslash, double-quote and
+   * null bytes.
+   *
+   * The callback may return either void or an error_code. When it returns an
+   * error_code, the walk stops at the first non-SUCCESS result and that error is
+   * returned, which lets the callback surface value-parse errors.
+   *
+   * This function is conditionally noexcept: it is noexcept exactly when invoking
+   * the callback is noexcept. The callback runs inside this frame, so a throwing
+   * callback (e.g. one using the exception-throwing conversions like
+   * std::string_view(value) or uint64_t(value)) makes for_each potentially
+   * throwing too -- the exception propagates to the caller instead of crossing a
+   * noexcept boundary and calling std::terminate.
+   *
+   * @returns a for_each_result holding the first error encountered while walking
+   *          the object (including any error returned by the callback, SUCCESS if
+   *          none) and the number of distinct selector keys that matched. The
+   *          result converts implicitly to error_code, so callers that only need
+   *          the error can ignore the count.
+   */
+  template <typename Selector, typename Func>
+    requires key_selector_type<Selector> &&
+             std::is_invocable_v<Func&, std::size_t, value>
+  simdjson_inline for_each_result for_each(Func&& on_match)
+      noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, value>);
+
+  /**
+   * Variadic per-key form. Provide exactly one handler per key in the Selector
+   * (compiler-enforced). Handlers are processed in JSON document order for the
+   * matching keys. Each handler is either:
+   *   - a deserialization target (a variable), in which case the matched value
+   *     is assigned to it via value::get -- no lambda required; or
+   *   - an invocable taking the ondemand::value (for custom logic such as
+   *     descending into a nested object). It may return void or error_code;
+   *     returning error_code lets you surface parse/type errors.
+   * The two styles may be mixed freely, one handler per key.
+   *
+   * Example (bind fields straight to variables):
+   *   using fields = ondemand::key_selector<"name", "city", "age">;
+   *   obj.for_each<fields>(name, city, age);
+   *
+   * Example (mixing a target and a lambda):
+   *   obj.for_each<ondemand::key_selector<"id", "user">>(
+   *     id,                                          // assigned via value::get
+   *     [&](ondemand::value v){ u = read_user(v); }  // custom logic
+   *   );
+   *
+   * The index-based single-callback form (taking (size_t, value)) remains
+   * available for shared-state or more complex per-key logic.
+   */
+  template <typename Selector, typename... Handlers>
+    requires key_selector_type<Selector> &&
+             (sizeof...(Handlers) == Selector::size()) &&
+             (key_selector_for_each_detail::field_handler<Handlers> && ...)
+  simdjson_inline for_each_result for_each(Handlers&&... on_match)
+      noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
+  /**
+   * Direct-key shorthand. Equivalent to for_each<key_selector<Keys...>>(...).
+   * Lets you write the keys inline without a separate using/alias, binding each
+   * field straight to a variable (or a lambda, see the Selector form above):
+   *
+   *   obj.for_each<"name", "city", "age">(name, city, age);
+   */
+  template <constevalutil::fixed_string... Keys, typename... Handlers>
+    requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+             (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+             (key_selector_for_each_detail::field_handler<Handlers> && ...)
+  simdjson_inline for_each_result for_each(Handlers&&... on_match)
+      noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+#endif
+
   /**
    * Get the value associated with the given JSON pointer. We use the RFC 6901
    * https://tools.ietf.org/html/rfc6901 standard, interpreting the current node
@@ -178431,6 +229089,34 @@ public:
    * @returns true if the object contains some elements (not empty)
    */
   inline simdjson_result<bool> reset() & noexcept;
+  /**
+   * Get an opaque token representing the object's current scanning position.
+   * Pass it to revert_position() to return to this exact point later, without
+   * paying the cost of a full reset() and re-scan from the beginning.
+   *
+   * A typical use is an optional field that may or may not be next: capture
+   * the position, attempt find_field(), and on NO_SUCH_FIELD, revert_position()
+   * instead of reset() so that fields already consumed are not rescanned.
+   *
+   * The returned token is only valid for this object, and only until it is
+   * reset() or the parser iterate()s a new document; using it after either
+   * is undefined behavior (see object_position).
+   *
+   * @returns An opaque position token.
+   */
+  simdjson_inline object_position get_current_position() const noexcept;
+  /**
+   * Return the object's scanning position to a snapshot previously obtained
+   * from get_current_position(). Unlike reset(), this does not rescan the
+   * object from the beginning: fields before the captured position remain
+   * consumed, and scanning resumes exactly where the snapshot was captured.
+   *
+   * @param position A snapshot previously returned by get_current_position(),
+   *        for this same object.
+   * @returns SUCCESS, or OUT_OF_ORDER_ITERATION if called during active
+   *          iteration (SIMDJSON_DEVELOPMENT_CHECKS builds only).
+   */
+  simdjson_inline error_code revert_position(object_position position) noexcept;
   /**
    * This method scans the beginning of the object and checks whether the
    * object is empty.
@@ -178476,7 +229162,7 @@ public:
    */
   template <typename T>
   simdjson_warn_unused simdjson_inline error_code get(T &out)
-     noexcept(custom_deserializable<T, object> ? nothrow_custom_deserializable<T, object> : true) {
+     noexcept(nothrow_gettable<T, object>) {
     static_assert(custom_deserializable<T, object>);
     return deserialize(*this, out);
   }
@@ -178488,7 +229174,7 @@ public:
    */
   template <typename T>
   simdjson_inline simdjson_result<T> get()
-    noexcept(custom_deserializable<T, value> ? nothrow_custom_deserializable<T, value> : true)
+    noexcept(nothrow_gettable<T, object>)
   {
     static_assert(std::is_default_constructible<T>::value, "The specified type is not default constructible.");
     T out{};
@@ -178540,10 +229226,18 @@ protected:
   simdjson_warn_unused simdjson_inline error_code find_field_raw(const std::string_view key) noexcept;

   value_iterator iter{};
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  bool locked{false};
+  simdjson_inline void set_locked(bool _locked) noexcept;
+#endif

   friend class value;
   friend class document;
   friend struct simdjson_result<object>;
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  friend class object_iterator;
+  friend struct simdjson_result<object_iterator>;
+#endif
 };

 } // namespace ondemand
@@ -178559,7 +229253,8 @@ public:
   simdjson_inline simdjson_result(error_code error) noexcept; ///< @private
   simdjson_inline simdjson_result() noexcept = default;

-  simdjson_inline simdjson_result<rvv_vls::ondemand::object_iterator> begin() noexcept;
+  simdjson_inline simdjson_result<rvv_vls::ondemand::object_iterator> begin() & noexcept;
+  simdjson_inline simdjson_result<rvv_vls::ondemand::object_iterator> begin() && noexcept;
   simdjson_inline simdjson_result<rvv_vls::ondemand::object_iterator> end() noexcept;
   simdjson_inline simdjson_result<rvv_vls::ondemand::value> find_field(std::string_view key) & noexcept;
   simdjson_inline simdjson_result<rvv_vls::ondemand::value> find_field(std::string_view key) && noexcept;
@@ -178577,6 +229272,8 @@ public:
 #endif
   simdjson_inline error_code for_each_at_path_with_wildcard(std::string_view json_path, Func&& callback) noexcept;
   inline simdjson_result<bool> reset() noexcept;
+  inline simdjson_result<rvv_vls::ondemand::object_position> get_current_position() noexcept;
+  inline error_code revert_position(rvv_vls::ondemand::object_position position) noexcept;
   inline simdjson_result<bool> is_empty() noexcept;
   inline simdjson_result<size_t> count_fields() & noexcept;
   inline simdjson_result<std::string_view> raw_json() noexcept;
@@ -178584,7 +229281,7 @@ public:
   // TODO: move this code into object-inl.h

   template<typename T>
-  simdjson_inline simdjson_result<T> get() noexcept {
+  simdjson_inline simdjson_result<T> get() noexcept(nothrow_gettable<T, rvv_vls::ondemand::object>) {
     if (error()) { return error(); }
     if constexpr (std::is_same_v<T, rvv_vls::ondemand::object>) {
       return first;
@@ -178592,7 +229289,7 @@ public:
     return first.get<T>();
   }
   template<typename T>
-  simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept {
+  simdjson_warn_unused simdjson_inline error_code get(T& out) noexcept(nothrow_gettable<T, rvv_vls::ondemand::object>) {
     if (error()) { return error(); }
     if constexpr (std::is_same_v<T, rvv_vls::ondemand::object>) {
       out = first;
@@ -178602,6 +229299,39 @@ public:
     return SUCCESS;
   }

+  /**
+   * Forwards to object::for_each on the underlying object, so error-code-style
+   * chains (e.g. doc["x"].get_object()) can call for_each without first
+   * extracting the object. If this result holds an error, that error is returned
+   * (with a zero match count) and the callback is not invoked. See
+   * object::for_each for the semantics.
+   */
+  template <typename Selector, typename Func>
+    requires rvv_vls::ondemand::key_selector_type<Selector> &&
+             std::is_invocable_v<Func&, std::size_t, rvv_vls::ondemand::value>
+  simdjson_inline rvv_vls::ondemand::for_each_result for_each(Func&& on_match)
+      noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, rvv_vls::ondemand::value>);
+
+  /**
+   * Forwarding overload for the variadic per-key form (targets and/or lambdas).
+   */
+  template <typename Selector, typename... Handlers>
+    requires rvv_vls::ondemand::key_selector_type<Selector> &&
+             (sizeof...(Handlers) == Selector::size()) &&
+             (rvv_vls::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+  simdjson_inline rvv_vls::ondemand::for_each_result for_each(Handlers&&... on_match)
+      noexcept(rvv_vls::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
+  /**
+   * Forwarding overload for the direct-key variadic form.
+   */
+  template <constevalutil::fixed_string... Keys, typename... Handlers>
+    requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+             (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+             (rvv_vls::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+  simdjson_inline rvv_vls::ondemand::for_each_result for_each(Handlers&&... on_match)
+      noexcept(rvv_vls::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>);
+
 #if SIMDJSON_STATIC_REFLECTION
   // TODO: move this code into object-inl.h
   template<constevalutil::fixed_string... FieldNames, typename T>
@@ -178642,6 +229372,15 @@ public:
    */
   simdjson_inline object_iterator() noexcept = default;

+#if SIMDJSON_DEVELOPMENT_CHECKS
+   simdjson_inline ~object_iterator() noexcept;
+
+   simdjson_inline object_iterator(object_iterator&&) noexcept;
+   simdjson_inline object_iterator& operator=(object_iterator&&) noexcept;
+   simdjson_inline object_iterator(const object_iterator&) noexcept;
+   simdjson_inline object_iterator& operator=(const object_iterator&) noexcept;
+#endif
+
   //
   // Iterator interface
   //
@@ -178661,6 +229400,9 @@ public:
 private:
 #if SIMDJSON_DEVELOPMENT_CHECKS
    bool has_been_referenced{false};
+   object* parent{nullptr};
+
+   simdjson_inline object_iterator(const value_iterator &_iter, object* _parent) noexcept;
 #endif
   /**
    * The underlying JSON iterator.
@@ -178706,6 +229448,191 @@ public:

 #endif // SIMDJSON_GENERIC_ONDEMAND_OBJECT_ITERATOR_H
 /* end file simdjson/generic/ondemand/object_iterator.h for rvv_vls */
+/* including simdjson/generic/ondemand/ranges.h for rvv_vls: #include "simdjson/generic/ondemand/ranges.h" */
+/* begin file simdjson/generic/ondemand/ranges.h for rvv_vls */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_RANGES_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_ONDEMAND_RANGES_H */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object_iterator.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/field.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#if SIMDJSON_SUPPORTS_RANGES
+
+namespace simdjson {
+namespace rvv_vls {
+namespace ondemand {
+
+/**
+ * A ranges-compatible iterator adapter for JSON arrays.
+ *
+ * Wraps array_iterator to satisfy std::input_iterator by providing:
+ * - const operator* (via mutable internal state)
+ * - post-increment operator
+ * - iterator_concept tag
+ *
+ * The mutable approach is standard for single-pass input iterators that
+ * read from external sources (similar to std::istream_iterator).
+ */
+class array_range_iterator {
+public:
+  using iterator_concept = std::input_iterator_tag;
+  using value_type = simdjson_result<value>;
+  using reference = simdjson_result<value>;
+  using difference_type = std::ptrdiff_t;
+
+  simdjson_inline array_range_iterator() noexcept = default;
+  simdjson_inline explicit array_range_iterator(array_iterator iter) noexcept;
+
+  /**
+   * Get the current element. Const-qualified for std::indirectly_readable;
+   * internally delegates to the mutable wrapped iterator.
+   */
+  simdjson_inline simdjson_result<value> operator*() const noexcept;
+  simdjson_inline array_range_iterator& operator++() noexcept;
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+  simdjson_inline void operator++(int) noexcept;
+SIMDJSON_POP_DISABLE_WARNINGS
+  /**
+   * Comparison delegates to array_iterator::operator==, which checks
+   * whether the underlying parser has finished the array (depth-based).
+   */
+  simdjson_inline friend bool operator==(const array_range_iterator& a,
+                                         const array_range_iterator& b) noexcept {
+    return a.iter_ == b.iter_;
+  }
+
+private:
+  mutable array_iterator iter_{};
+};
+
+/**
+ * A std::ranges::view over a JSON array.
+ *
+ * Wraps an ondemand::array and exposes begin()/end() that return
+ * array_range_iterator (satisfying std::input_iterator), enabling
+ * use with std::views::transform and other range adaptors.
+ *
+ * If the array's begin() returns an error (only possible under
+ * SIMDJSON_DEVELOPMENT_CHECKS), the range will be empty and error()
+ * will return the error code.
+ *
+ * Usage:
+ *   ondemand::parser parser;
+ *   auto doc = parser.iterate(json);
+ *   auto arr = doc.get_array().value();
+ *   for (auto elem : ondemand::get_range(arr)) { ... }
+ */
+class array_range {
+public:
+  simdjson_inline array_range() noexcept = default;
+  simdjson_inline explicit array_range(array& arr) noexcept;
+
+  simdjson_inline array_range_iterator begin() noexcept;
+  simdjson_inline array_range_iterator end() noexcept;
+
+  /** Returns SUCCESS if the range was created successfully, or the error code otherwise. */
+  simdjson_inline error_code error() const noexcept { return error_; }
+
+private:
+  array_iterator begin_{};
+  array_iterator end_{};
+  error_code error_{SUCCESS};
+};
+
+/**
+ * A ranges-compatible iterator adapter for JSON objects.
+ *
+ * Wraps object_iterator to satisfy std::input_iterator, yielding
+ * simdjson_result<field> elements (key-value pairs).
+ */
+class object_range_iterator {
+public:
+  using iterator_concept = std::input_iterator_tag;
+  using value_type = simdjson_result<field>;
+  using reference = simdjson_result<field>;
+  using difference_type = std::ptrdiff_t;
+
+  simdjson_inline object_range_iterator() noexcept = default;
+  simdjson_inline explicit object_range_iterator(object_iterator iter) noexcept;
+
+  simdjson_inline simdjson_result<field> operator*() const noexcept;
+  simdjson_inline object_range_iterator& operator++() noexcept;
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+  simdjson_inline void operator++(int) noexcept;
+SIMDJSON_POP_DISABLE_WARNINGS
+  simdjson_inline friend bool operator==(const object_range_iterator& a,
+                                         const object_range_iterator& b) noexcept {
+    return a.iter_ == b.iter_;
+  }
+
+private:
+  mutable object_iterator iter_{};
+};
+
+/**
+ * A std::ranges::view over a JSON object.
+ *
+ * Wraps an ondemand::object and exposes begin()/end() that return
+ * object_range_iterator, enabling use with range adaptors.
+ *
+ * If the object's begin() returns an error, the range will be empty
+ * and error() will return the error code.
+ */
+class object_range {
+public:
+  simdjson_inline object_range() noexcept = default;
+  simdjson_inline explicit object_range(object& obj) noexcept;
+
+  simdjson_inline object_range_iterator begin() noexcept;
+  simdjson_inline object_range_iterator end() noexcept;
+
+  /** Returns SUCCESS if the range was created successfully, or the error code otherwise. */
+  simdjson_inline error_code error() const noexcept { return error_; }
+
+private:
+  object_iterator begin_{};
+  object_iterator end_{};
+  error_code error_{SUCCESS};
+};
+
+/** Get a std::ranges compatible view over a JSON array. */
+simdjson_inline array_range get_range(array& arr) noexcept;
+
+/** Get a std::ranges compatible view over a JSON object (key-value pairs). */
+simdjson_inline object_range get_key_value_range(object& obj) noexcept;
+
+#if SIMDJSON_EXCEPTIONS
+/** Get a std::ranges compatible view, unwrapping the simdjson_result (throws on error). */
+simdjson_inline array_range get_range(simdjson_result<array> result);
+
+/** Get a std::ranges compatible view, unwrapping the simdjson_result (throws on error). */
+simdjson_inline object_range get_key_value_range(simdjson_result<object> result);
+#endif // SIMDJSON_EXCEPTIONS
+
+} // namespace ondemand
+} // namespace rvv_vls
+} // namespace simdjson
+
+namespace std {
+namespace ranges {
+template<>
+inline constexpr bool enable_view<simdjson::rvv_vls::ondemand::array_range> = true;
+template<>
+inline constexpr bool enable_view<simdjson::rvv_vls::ondemand::object_range> = true;
+} // namespace ranges
+} // namespace std
+
+#endif // SIMDJSON_SUPPORTS_RANGES
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_RANGES_H
+/* end file simdjson/generic/ondemand/ranges.h for rvv_vls */
 /* including simdjson/generic/ondemand/serialization.h for rvv_vls: #include "simdjson/generic/ondemand/serialization.h" */
 /* begin file simdjson/generic/ondemand/serialization.h for rvv_vls */
 #ifndef SIMDJSON_GENERIC_ONDEMAND_SERIALIZATION_H
@@ -178838,12 +229765,14 @@ inline std::ostream& operator<<(std::ostream& out, simdjson::simdjson_result<sim
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/annotations.h" */
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */

 #include <concepts>
 #include <limits>
 #if SIMDJSON_STATIC_REFLECTION
 #include <meta>
+#include <vector>
 // #include <static_reflection> // for std::define_static_string - header not available yet
 #endif

@@ -178868,10 +229797,25 @@ error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept {

 template <std::floating_point T>
 error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept {
-  double x;
-  SIMDJSON_TRY(val.get_double().get(x));
-  out = static_cast<T>(x);
-  return SUCCESS;
+  if constexpr (std::is_same_v<T, float>) {
+    // Going through binary64 and then rounding to binary32 would round twice
+    // and could produce a value that is not the float nearest to the JSON
+    // number, so we parse to binary32 directly.
+    return val.get_float().get(out);
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+  } else if constexpr (std::is_same_v<T, std::float32_t>) {
+    // Same reason as float.
+    float x;
+    SIMDJSON_TRY(val.get_float().get(x));
+    out = static_cast<T>(x);
+    return SUCCESS;
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+  } else {
+    double x;
+    SIMDJSON_TRY(val.get_double().get(x));
+    out = static_cast<T>(x);
+    return SUCCESS;
+  }
 }

 template <std::signed_integral T>
@@ -178907,11 +229851,59 @@ template <concepts::constructible_from_string_view T, typename ValT>
 error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(std::is_nothrow_constructible_v<T, std::string_view>) {
   std::string_view str;
   SIMDJSON_TRY(val.get_string().get(str));
-  out = T{str};
+  if constexpr (requires { out.assign(str.data(), str.size()); }) {
+    // Copy straight into out (e.g., std::string): building a temporary and
+    // move-assigning it is markedly slower.
+    out.assign(str.data(), str.size());
+  } else {
+    out = T{str};
+  }
+  return SUCCESS;
+}
+
+#if SIMDJSON_SUPPORTS_CHAR8_T
+// any C++20 char8_t string-like type (can be constructed from std::u8string_view),
+// such as std::u8string
+template <concepts::constructible_from_u8string_view T, typename ValT>
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(std::is_nothrow_constructible_v<T, std::u8string_view>) {
+  std::u8string_view str;
+  SIMDJSON_TRY(val.get_u8string().get(str));
+  if constexpr (requires { out.assign(str.data(), str.size()); }) {
+    // Copy straight into out (e.g., std::u8string), as for std::string above.
+    out.assign(str.data(), str.size());
+  } else {
+    out = T{str};
+  }
   return SUCCESS;
 }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T


+namespace details {
+// Whether to deserialize the elements of the container T directly into a new
+// element (emplace_one(out) and then get) rather than into a temporary that is
+// then moved into the container. A temporary of a trivially copyable type lives
+// in registers and is cheap to move, and value-initializing a small element in
+// place costs more than it saves (GCC zeroes it with rep stos). But moving a
+// large element, or a string (whose deserialization then needs a temporary
+// string and a move assignment of its own), is expensive.
+template <typename T>
+concept deserialize_in_place =
+    concepts::returns_reference<T> && requires(T &c) { c.pop_back(); } &&
+    !std::is_trivially_copyable_v<typename T::value_type> &&
+    (sizeof(typename T::value_type) > 32 || concepts::constructible_from_string_view<typename T::value_type>);
+
+// Removes the last element of the container on destruction while armed.
+template <typename T>
+struct pop_back_guard {
+  T &container;
+  bool armed{true};
+  ~pop_back_guard() {
+    if (armed) { container.pop_back(); }
+  }
+};
+} // namespace details
+
 /**
  * STL containers have several constructors including one that takes a single
  * size argument. Thus, some compilers (Visual Studio) will not be able to
@@ -178935,22 +229927,65 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
     SIMDJSON_TRY(val.get_array().get(arr));
   }

-  for (auto v : arr) {
-    if constexpr (concepts::returns_reference<T>) {
-      if (auto const err = v.get<value_type>().get(concepts::emplace_one(out));
-          err) {
-        // If an error occurs, the empty element that we just inserted gets
-        // removed. We're not using a temp variable because if T is a heavy
-        // type, we want the valid path to be the fast path and the slow path be
-        // the path that has errors in it.
-        if constexpr (requires { out.pop_back(); }) {
-          static_cast<void>(out.pop_back());
+  if constexpr (std::is_same_v<T, std::vector<value_type>> && !std::is_same_v<value_type, bool>) {
+    // Collect the elements in a per-thread scratch vector that keeps its
+    // capacity from call to call, then move them into out after reserving the
+    // exact size: out is allocated once instead of being regrown. A nested
+    // array of the same type finds the scratch busy and takes the paths below.
+    // Prior related work: jsonifier keeps a thread-local vector and sizes the
+    // caller's vector from that element count (parse_impl.hpp,
+    // https://github.com/nihilai-collective/Jsonifier).
+    struct scratch_space {
+      std::vector<value_type> elements{};
+      bool busy{false};
+    };
+    static thread_local scratch_space scratch;
+    if (!scratch.busy && out.empty()) {
+      struct release_scratch {
+        scratch_space &s;
+        T &out;
+        size_t parsed{0};
+        bool complete{false};
+        // On an error or an exception, out gets the elements parsed so far (as
+        // with the loops below), without allocating. Kept out of the hot path.
+        simdjson_never_inline void keep_parsed() noexcept {
+          s.elements.resize(parsed);
+          out.swap(s.elements);
         }
-        return err;
-      }
-    } else {
+        ~release_scratch() {
+          if (simdjson_unlikely(!complete)) { keep_parsed(); }
+          s.elements.clear();
+          // Do not hold on to the memory of a very large array.
+          if (s.elements.capacity() * sizeof(value_type) > (1 << 20)) { std::vector<value_type>().swap(s.elements); }
+          s.busy = false;
+        }
+      } release{scratch, out};
+      scratch.busy = true;
+      for (auto v : arr) {
+        SIMDJSON_TRY(v.get<value_type>(scratch.elements.emplace_back()));
+        release.parsed++;
+      }
+      out.reserve(release.parsed);
+      release.complete = true;
+      for (auto &e : scratch.elements) { out.emplace_back(std::move(e)); }
+      return SUCCESS;
+    }
+  }
+  if constexpr (details::deserialize_in_place<T>) {
+    for (auto v : arr) {
+      auto &slot = concepts::emplace_one(out);
+      // An error or an exception (a user tag_invoke may throw) must not leave
+      // a partially deserialized element behind.
+      details::pop_back_guard<T> guard{out};
+      SIMDJSON_TRY(v.get<value_type>(slot));
+      guard.armed = false;
+    }
+  } else {
+    for (auto v : arr) {
+      // Deserialize into a temporary first: an error or an exception (a user
+      // tag_invoke may throw) must not leave a default-constructed element behind.
       value_type temp;
-      if (auto const err = v.get<value_type>().get(temp); err) {
+      if (auto const err = v.get<value_type>(temp); err) {
         return err;
       }
       concepts::emplace_one(out, std::move(temp));
@@ -178991,7 +230026,7 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
 }

 template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, rvv_vls::ondemand::object &obj, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, rvv_vls::ondemand::object &obj, T &out) noexcept(false) {
   using value_type = typename std::remove_cvref_t<T>::mapped_type;

   out.clear();
@@ -179010,21 +230045,21 @@ error_code tag_invoke(deserialize_tag, rvv_vls::ondemand::object &obj, T &out) n
 }

 template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, rvv_vls::ondemand::value &val, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, rvv_vls::ondemand::value &val, T &out) noexcept(false) {
   rvv_vls::ondemand::object obj;
   SIMDJSON_TRY(val.get_object().get(obj));
   return simdjson::deserialize(obj, out);
 }

 template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, rvv_vls::ondemand::document &doc, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, rvv_vls::ondemand::document &doc, T &out) noexcept(false) {
   rvv_vls::ondemand::object obj;
   SIMDJSON_TRY(doc.get_object().get(obj));
   return simdjson::deserialize(obj, out);
 }

 template <concepts::string_view_keyed_map T>
-error_code tag_invoke(deserialize_tag, rvv_vls::ondemand::document_reference &doc, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, rvv_vls::ondemand::document_reference &doc, T &out) noexcept(false) {
   rvv_vls::ondemand::object obj;
   SIMDJSON_TRY(doc.get_object().get(obj));
   return simdjson::deserialize(obj, out);
@@ -179035,10 +230070,6 @@ error_code tag_invoke(deserialize_tag, rvv_vls::ondemand::document_reference &do
  * This CPO (Customization Point Object) will help deserialize into
  * smart pointers.
  *
- * If constructing T is nothrow, this conversion should be nothrow as well since
- * we return MEMALLOC if we're not able to allocate memory instead of throwing
- * the error message.
- *
  * @tparam T The type inside the smart pointer
  * @tparam ValT document/value type
  * @param val document/value
@@ -179046,7 +230077,7 @@ error_code tag_invoke(deserialize_tag, rvv_vls::ondemand::document_reference &do
  * @return status of the conversion
  */
 template <concepts::smart_pointer T, typename ValT>
-error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(nothrow_deserializable<typename std::remove_cvref_t<T>::element_type, ValT>) {
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
   using element_type = typename std::remove_cvref_t<T>::element_type;

   // For better error messages, don't use these as constraints on
@@ -179058,12 +230089,13 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(nothrow_deser
       std::is_default_constructible_v<element_type>,
       "The specified type inside the unique_ptr must default constructible.");

-  auto ptr = new (std::nothrow) element_type();
-  if (ptr == nullptr) {
+  // Own the allocation before get(): a user tag_invoke may throw.
+  std::unique_ptr<element_type> ptr(new (std::nothrow) element_type());
+  if (!ptr) {
     return MEMALLOC;
   }
   SIMDJSON_TRY(val.template get<element_type>(*ptr));
-  out.reset(ptr);
+  out = std::move(ptr);
   return SUCCESS;
 }

@@ -179095,53 +230127,595 @@ error_code tag_invoke(deserialize_tag, auto &val, T &out) noexcept(nothrow_deser

 template <typename T>
 constexpr bool user_defined_type = (std::is_class_v<T>
-&& !std::is_same_v<T, std::string> && !std::is_same_v<T, std::string_view> && !concepts::optional_type<T> &&
-!concepts::appendable_containers<T>);
+&& !std::is_same_v<T, std::string> && !std::is_same_v<T, std::string_view>
+#if SIMDJSON_SUPPORTS_CHAR8_T
+// The char8_t string types are class types with no reflectable members, so
+// without this they would be taken for user structs and the reflection
+// overload below would compete with the u8 string overload in
+// std_deserialize.h, making every tag_invoke call on them ambiguous.
+&& !std::is_same_v<T, std::u8string> && !std::is_same_v<T, std::u8string_view>
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+&& !concepts::optional_type<T> &&
+!concepts::appendable_containers<T>
+// simdjson's own types (array, object, value, raw_json_string, number,
+// document, document_reference) have dedicated get<T>() specializations and
+// must never go through reflection.
+&& !is_builtin_deserializable_v<T>
+&& !std::is_same_v<T, rvv_vls::ondemand::number>
+&& !std::is_same_v<T, rvv_vls::ondemand::document>
+&& !std::is_same_v<T, rvv_vls::ondemand::document_reference>);
+
+
+// key_selector_reflection_detail is defined unconditionally (it only requires
+// static reflection). It provides the compile-time machinery for building a
+// key_selector from a struct's members, the per-member helpers shared by every
+// deserialization path (they implement the annotations of annotations.h), the
+// ordered per-member path used by the opt-out build (see
+// deserialize_struct_ordered below), and the scan used for structs annotated
+// with deny_unknown_fields and as an automatic fallback (see
+// deserialize_struct_scan and keys_fit_selector below).
+namespace key_selector_reflection_detail {
+
+// A member participates if it is public, non-const, and not annotated with skip
+// or skip_deserializing.
+consteval bool is_eligible_member(std::meta::info mem) {
+  return !std::meta::is_const(mem) && std::meta::is_public(mem)
+      && !simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_tag)
+      && !simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_deserializing_tag);
+}
+
+// A field is a data member reached from T through a chain of members: usually a
+// direct member of T (a path of length one), or a member of a member annotated
+// with flatten (the members of a flattened structure are fields of the enclosing
+// one). A field is identified by the type member_path<m1, m2, ..., leaf>.
+template <std::meta::info First, std::meta::info... Rest>
+struct member_path {
+  // The data member holding the value; its annotations drive (de)serialization.
+  static constexpr std::meta::info leaf = [] {
+    std::meta::info members[] = {First, Rest...};
+    return members[sizeof...(Rest)];
+  }();
+  template <typename T>
+  static simdjson_inline constexpr auto &get(T &obj) noexcept {
+    if constexpr (sizeof...(Rest) == 0) {
+      return obj.[:First:];
+    } else {
+      return member_path<Rest...>::get(obj.[:First:]);
+    }
+  }
+};
+
+// True when T can be flattened: a structure deserialized member by member (not
+// a string, a container, an optional, a smart pointer, ...).
+template <typename T>
+constexpr bool flattenable_type = user_defined_type<T> && !concepts::string_view_keyed_map<T>
+    && !concepts::container_but_not_string<T> && !concepts::smart_pointer<T>;
+
+consteval void append_eligible_fields(std::meta::info type, std::vector<std::meta::info> &prefix,
+                                      std::vector<std::meta::info> &fields) {
+  for (std::meta::info mem : std::meta::nonstatic_data_members_of(type, std::meta::access_context::unchecked())) {
+    if (!is_eligible_member(mem)) { continue; }
+    prefix.push_back(std::meta::reflect_constant(mem));
+    if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::flatten_tag)) {
+      std::meta::info flattened = simdjson::detail::flattened_type(mem);
+      if (!std::meta::extract<bool>(std::meta::substitute(^^flattenable_type, {flattened}))) {
+        throw std::meta::exception(u8"simdjson::flatten requires a member whose type is a structure deserialized member by member", mem);
+      }
+      append_eligible_fields(flattened, prefix, fields);
+    } else {
+      fields.push_back(std::meta::substitute(^^member_path, prefix));
+    }
+    prefix.pop_back();
+  }
+}
+
+// The fields of `type` that participate in deserialization, in declaration order
+// (the fields of a flattened member take its place). A field's position in this
+// list is its "field index" below.
+consteval std::vector<std::meta::info> eligible_fields(std::meta::info type) {
+  std::vector<std::meta::info> prefix;
+  std::vector<std::meta::info> fields;
+  append_eligible_fields(type, prefix, fields);
+  return fields;
+}
+
+// The data member at the end of a field path.
+consteval std::meta::info field_leaf(std::meta::info path) {
+  return std::meta::extract<std::meta::info>(std::meta::template_arguments_of(path).back());
+}
+
+// Number of fields that participate in deserialization. A class can have zero
+// eligible fields (e.g. std::chrono::time_point, whose only data member is
+// private): an empty key_selector cannot be built, so the tag_invoke below
+// special-cases this count.
+template <typename T>
+consteval std::size_t eligible_field_count() {
+  return eligible_fields(^^T).size();
+}
+
+// The keys accepted for a member (or an enumerator): its JSON key followed by its
+// aliases. They are static strings, so they can be used as template arguments.
+consteval std::vector<const char *> accepted_keys_of(std::meta::info entity) {
+  std::vector<const char *> keys;
+  for (std::string_view key : simdjson::detail::json_key_names(entity)) {
+    bool repeated = false;
+    for (const char *previous : keys) { repeated = repeated || std::string_view(previous) == key; }
+    if (!repeated) { keys.push_back(std::define_static_string(key)); }
+  }
+  return keys;
+}
+
+// Every key accepted for T: the eligible fields in order, each followed by its
+// aliases. This is the key list of T's key_selector.
+consteval std::vector<const char *> accepted_keys(std::meta::info type) {
+  std::vector<const char *> keys;
+  for (std::meta::info path : eligible_fields(type)) {
+    for (const char *key : accepted_keys_of(field_leaf(path))) { keys.push_back(key); }
+  }
+  return keys;
+}
+
+// The field index of each entry of accepted_keys(type).
+consteval std::vector<std::size_t> accepted_key_fields(std::meta::info type) {
+  std::vector<std::size_t> key_fields;
+  std::vector<std::meta::info> fields = eligible_fields(type);
+  for (std::size_t i = 0; i < fields.size(); ++i) {
+    for (std::size_t j = 0; j < accepted_keys_of(field_leaf(fields[i])).size(); ++j) { key_fields.push_back(i); }
+  }
+  return key_fields;
+}
+
+// True when no two eligible fields of T accept the same key. Otherwise one JSON
+// key would have to fill several members: this is reported at compile time.
+template <typename T>
+consteval bool accepted_keys_are_distinct() {
+  std::vector<const char *> keys = accepted_keys(^^T);
+  for (std::size_t i = 0; i < keys.size(); ++i) {
+    for (std::size_t j = i + 1; j < keys.size(); ++j) {
+      if (std::string_view(keys[i]) == std::string_view(keys[j])) { return false; }
+    }
+  }
+  return true;
+}
+
+// True when some accepted key of T is written with escape sequences in JSON (a
+// double quote, a backslash or a control character). Such a key can only be
+// matched by comparing unescaped keys (deserialize_struct_scan): obj[key] and
+// the key_selector compare the raw bytes.
+template <typename T>
+consteval bool keys_need_unescaping() {
+  for (std::string_view key : accepted_keys(^^T)) {
+    for (char c : key) {
+      if (c == '\\' || c == '"' || static_cast<unsigned char>(c) < 0x20) { return true; }
+    }
+  }
+  return false;
+}

+// True when some eligible field of T has aliases: several selector keys may then
+// map to the same field.
+template <typename T>
+consteval bool has_aliases() {
+  return accepted_keys(^^T).size() != eligible_fields(^^T).size();
+}
+
+// True when `mem` is annotated with default_value or default_from (directly, or
+// through a default_value annotation on the enclosing structure).
+template <auto mem>
+consteval bool has_default() {
+  return simdjson::detail::has_annotation(mem, ^^simdjson::detail::default_value_tag)
+      || simdjson::detail::has_annotation(std::meta::parent_of(mem), ^^simdjson::detail::default_value_tag)
+      || simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::default_from_t) != std::meta::info{};
+}
+
+// True when a missing key for `mem` is not an error: optional members and
+// members with a default.
+template <auto mem>
+consteval bool may_be_absent() {
+  return concepts::optional_type<typename [: std::meta::type_of(mem) :]> || has_default<mem>();
+}
+
+// True when every eligible field of T is required (none may be absent). In that
+// case presence can be checked with a single match count instead of a per-field
+// "seen" array.
+template <typename T>
+consteval bool all_eligible_fields_required() {
+  bool all_required = true;
+  template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+    if constexpr (may_be_absent<[: path :]::leaf>()) {
+      all_required = false;
+    }
+  }
+  return all_required;
+}
+
+// `key` as a constevalutil::fixed_string usable as an NTTP.
+template <const char *key>
+consteval auto key_fixed_string() {
+  constexpr std::string_view key_view{ key };
+  char buffer[key_view.size() + 1] = {};
+  for (std::size_t i = 0; i < key_view.size(); ++i) { buffer[i] = key_view[i]; }
+  return constevalutil::fixed_string<key_view.size() + 1>(buffer);
+}
+
+// key_selector template arguments (one fixed_string per accepted key), in the
+// order of accepted_keys.
+template <typename T>
+consteval std::vector<std::meta::info> selector_key_args() {
+  std::vector<std::meta::info> args;
+  template for (constexpr const char *key : std::define_static_array(accepted_keys(^^T))) {
+    args.push_back(std::meta::reflect_constant(key_fixed_string<key>()));
+  }
+  return args;
+}
+
+// key_selector whose keys are exactly T's accepted keys (index i <-> i-th key).
+template <typename T>
+using selector_for = typename [: std::meta::substitute(
+    ^^rvv_vls::ondemand::key_selector, selector_key_args<T>()) :];
+
+// True when T's accepted keys satisfy every key_selector requirement, so a
+// key_selector can be built for T without a compile-time error. This mirrors the
+// key_selector limits (see key_selector.h): at most 255 keys, each key non-empty
+// and at most 63 characters, no backslash / double-quote / null byte, and all
+// keys distinct. Keys with any other control character are excluded too: they
+// are escaped in JSON, so their raw bytes never match. When this returns false
+// the deserializer falls back to deserialize_struct_scan instead of failing to
+// compile.
+template <typename T>
+consteval bool keys_fit_selector() {
+  std::vector<std::string_view> keys;
+  for (const char *key : accepted_keys(^^T)) { keys.push_back(key); }
+  if (keys.size() > 255) { return false; }
+  for (std::size_t i = 0; i < keys.size(); ++i) {
+    if (keys[i].empty() || keys[i].size() > 63) { return false; }
+    for (char c : keys[i]) {
+      if (c == '\\' || c == '"' || static_cast<unsigned char>(c) < 0x20) { return false; }
+    }
+    for (std::size_t j = i + 1; j < keys.size(); ++j) {
+      if (keys[i] == keys[j]) { return false; }
+    }
+  }
+  return true;
+}
+
+// True when the class `adapter` declares a member named deserialize.
+consteval bool declares_deserialize(std::meta::info adapter) {
+  for (std::meta::info m : std::meta::members_of(adapter, std::meta::access_context::unchecked())) {
+    if (std::meta::has_identifier(m) && std::meta::identifier_of(m) == "deserialize") { return true; }
+  }
+  return false;
+}
+
+// Deserialize a JSON value into `target`, the storage of member `mem`, through
+// the member's with<Adapter> annotation when the adapter provides a deserialize
+// function.
+template <auto mem, typename ValueT, typename M>
+simdjson_warn_unused simdjson_inline error_code deserialize_member_value(ValueT &field_value, M &target) noexcept(false) {
+  constexpr std::meta::info with_type = simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::with_t);
+  if constexpr (with_type != std::meta::info{}) {
+    using adapter = typename [: with_type :]::adapter;
+    using ondemand_value = rvv_vls::ondemand::value;
+    if constexpr (requires { { adapter::deserialize(field_value, target) } -> std::convertible_to<error_code>; }) {
+      return adapter::deserialize(field_value, target);
+    } else if constexpr (requires(ondemand_value &v) { { adapter::deserialize(v, target) } -> std::convertible_to<error_code>; }
+                         && requires { field_value.get_value(); }) {
+      // A transparent structure read from a document: the adapter takes an
+      // ondemand::value. A scalar document cannot be viewed as a value, so it
+      // reports SCALAR_DOCUMENT_AS_VALUE (an adapter taking auto& receives the
+      // document itself and has no such limitation).
+      ondemand_value v;
+      SIMDJSON_TRY(field_value.get_value().get(v));
+      return adapter::deserialize(v, target);
+    } else {
+      static_assert(!declares_deserialize(^^adapter),
+                    "the deserialize function of a simdjson::with adapter must be callable as "
+                    "Adapter::deserialize(simdjson::ondemand::value &, T &) and return an error_code");
+      return field_value.get(target);
+    }
+  } else {
+    return field_value.get(target);
+  }
+}
+
+// Deserialize the storage `target` of member `mem` from a JSON value.
+template <auto mem, typename ValueT, typename M>
+simdjson_warn_unused simdjson_inline error_code deserialize_member(ValueT &field_value, M &target) noexcept(false) {
+  if constexpr (has_default<mem>() && std::is_default_constructible_v<M> && std::is_move_assignable_v<M>) {
+    // A present key replaces the default value: deserialize into a fresh
+    // temporary so that, e.g., a container does not append to its default
+    // content, and a failure leaves the default untouched.
+    M value{};
+    SIMDJSON_TRY(deserialize_member_value<mem>(field_value, value));
+    target = std::move(value);
+    return SUCCESS;
+  } else {
+    return deserialize_member_value<mem>(field_value, target);
+  }
+}

+// Deserialize the field with the given field index.
+template <typename T>
+simdjson_warn_unused simdjson_inline error_code deserialize_field_at(
+    std::size_t field_index, rvv_vls::ondemand::value field_value, T &out) noexcept(false) {
+  std::size_t counter = 0;
+  template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+    using field = [: path :];
+    if (field_index == counter) { return deserialize_member<field::leaf>(field_value, field::get(out)); }
+    ++counter;
+  }
+  return SUCCESS;
+}
+
+// Called when the key(s) of member `mem`, stored in `target`, are missing from
+// the JSON object: an error for a required member, a call to the default_from
+// factory, or nothing at all.
+template <auto mem, typename M>
+simdjson_warn_unused simdjson_inline error_code on_missing_member(M &target) noexcept(false) {
+  constexpr std::meta::info default_from_type = simdjson::detail::annotation_of_template(mem, ^^simdjson::detail::default_from_t);
+  if constexpr (default_from_type != std::meta::info{}) {
+    target = [: default_from_type :]::factory();
+    return SUCCESS;
+  } else if constexpr (may_be_absent<mem>()) {
+    // For optional and default_value members, a missing key is not an error:
+    // leave the member at its current (default) value.
+    (void)target;
+    return SUCCESS;
+  } else {
+    (void)target;
+    return NO_SUCH_FIELD;
+  }
+}
+
+// Report or handle every field that was not seen in the JSON object.
+template <typename T, std::size_t N>
+simdjson_warn_unused simdjson_inline error_code handle_missing_fields(
+    const std::array<bool, N> &seen_field, T &out) noexcept(false) {
+  std::size_t counter = 0;
+  template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+    using field = [: path :];
+    if (!seen_field[counter]) { SIMDJSON_TRY(on_missing_member<field::leaf>(field::get(out))); }
+    ++counter;
+  }
+  return SUCCESS;
+}
+
+// Ordered, per-field deserialization: one obj[key] lookup per eligible field
+// (and per alias, until one is found). This is the opt-out path
+// (-DSIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION=1), except for structs with keys
+// that need unescaping (see keys_need_unescaping).
+template <typename T>
+simdjson_warn_unused error_code deserialize_struct_ordered(
+    rvv_vls::ondemand::object &obj, T &out) noexcept(false) {
+  template for (constexpr auto path : std::define_static_array(eligible_fields(^^T))) {
+    using field = [: path :];
+    rvv_vls::ondemand::value field_value;
+    error_code error = NO_SUCH_FIELD;
+    template for (constexpr const char *key : std::define_static_array(accepted_keys_of(field::leaf))) {
+      if (error == NO_SUCH_FIELD) { error = obj[std::string_view(key)].get(field_value); }
+    }
+    if (error == NO_SUCH_FIELD) {
+      SIMDJSON_TRY(on_missing_member<field::leaf>(field::get(out)));
+    } else if (error) {
+      return error;
+    } else {
+      SIMDJSON_TRY(deserialize_member<field::leaf>(field_value, field::get(out)));
+    }
+  }
+  return SUCCESS;
+}
+
+// Appends the JSON keys that serialization writes for members of `type` that
+// deserialization cannot assign (const or non-public members, directly or
+// through flatten). `all` is true inside a flattened member that is itself
+// unassignable. Keys of skip_deserializing members are not included: they are
+// unknown keys, as documented.
+consteval void append_unassignable_keys(std::meta::info type, bool all, std::vector<const char *> &keys) {
+  for (std::meta::info mem : std::meta::nonstatic_data_members_of(type, std::meta::access_context::unchecked())) {
+    if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_tag)
+        || simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_serializing_tag)
+        || simdjson::detail::has_annotation(mem, ^^simdjson::detail::skip_deserializing_tag)) {
+      continue;
+    }
+    bool unassignable = all || !is_eligible_member(mem);
+    if (simdjson::detail::has_annotation(mem, ^^simdjson::detail::flatten_tag)) {
+      append_unassignable_keys(simdjson::detail::flattened_type(mem), unassignable, keys);
+    } else if (unassignable) {
+      keys.push_back(std::define_static_string(simdjson::detail::json_key_name(mem)));
+    }
+  }
+}
+
+consteval std::vector<const char *> unassignable_keys(std::meta::info type) {
+  std::vector<const char *> keys;
+  append_unassignable_keys(type, false, keys);
+  return keys;
+}
+
+// Deserialization by a single pass over every field of the object, comparing
+// unescaped keys. It is used for structs annotated with deny_unknown_fields
+// (DenyUnknown = true), where a key that does not map to an eligible field is
+// reported as UNKNOWN_FIELD, and as the fallback for structs whose keys the
+// key_selector or obj[key] cannot match (see keys_fit_selector and
+// keys_need_unescaping). As with object::for_each, the first occurrence of a
+// field wins (later duplicates, or aliases of a field already seen, are ignored).
+template <bool DenyUnknown, typename T>
+simdjson_warn_unused error_code deserialize_struct_scan(
+    rvv_vls::ondemand::object &obj, T &out) noexcept(false) {
+  static constexpr auto keys = std::define_static_array(accepted_keys(^^T));
+  static constexpr auto key_fields = std::define_static_array(accepted_key_fields(^^T));
+  std::array<bool, eligible_field_count<T>()> seen_field{};
+  for (auto field_result : obj) {
+    rvv_vls::ondemand::field json_field;
+    SIMDJSON_TRY(std::move(field_result).get(json_field));
+    std::string_view key;
+    SIMDJSON_TRY(json_field.unescaped_key().get(key));
+    std::size_t key_index = keys.size();
+    for (std::size_t i = 0; i < keys.size(); ++i) {
+      if (key == std::string_view(keys[i])) { key_index = i; break; }
+    }
+    if (key_index == keys.size()) {
+      if constexpr (DenyUnknown) {
+        // A key that T itself serializes (e.g. of a const member) is not
+        // unknown: a serialized value must parse back.
+        static constexpr auto ignored_keys = std::define_static_array(unassignable_keys(^^T));
+        bool ignored = false;
+        for (const char *ignored_key : ignored_keys) {
+          if (key == std::string_view(ignored_key)) { ignored = true; break; }
+        }
+        if (!ignored) { return UNKNOWN_FIELD; }
+      }
+      continue;
+    }
+    const std::size_t field_index = key_fields[key_index];
+    if (seen_field[field_index]) { continue; }
+    seen_field[field_index] = true;
+    SIMDJSON_TRY(deserialize_field_at(field_index, json_field.value(), out));
+  }
+  return handle_missing_fields(seen_field, out);
+}
+
+template <typename T>
+consteval bool is_transparent() {
+  return simdjson::detail::has_annotation(^^T, ^^simdjson::detail::transparent_tag);
+}
+
+} // namespace key_selector_reflection_detail
+
+// Deserialize a reflected struct. By default this builds a compile-time
+// key_selector from the struct's members and walks each object once with
+// object::for_each (perfect-hash key matching), instead of one obj[key] lookup
+// per member. Other paths are used instead:
+//   - globally, the ordered per-member path when defining
+//     -DSIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION=1;
+//   - automatically and per-type, a scan of the object comparing unescaped keys
+//     when the struct's keys do not fit the key_selector limits (see
+//     keys_fit_selector) or cannot be compared raw (see keys_need_unescaping),
+//     so that long member names and the like keep compiling rather than
+//     tripping a static_assert.
+// Structs annotated with deny_unknown_fields always use the strict scan, and
+// structs annotated with transparent are deserialized as their single member.
+//
+// noexcept(false): a member's tag_invoke may throw and the exception must reach
+// the caller of get<T>().
 template <typename T, typename ValT>
   requires(user_defined_type<T> && std::is_class_v<T>)
-error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
+error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept(false) {
+  if constexpr (key_selector_reflection_detail::is_transparent<T>()) {
+    constexpr auto mem = simdjson::detail::transparent_member(^^T);
+    if constexpr (std::is_same_v<std::remove_cvref_t<ValT>, rvv_vls::ondemand::object>) {
+      // We were handed an object: only a structure can be deserialized from it.
+      if constexpr (user_defined_type<typename [: std::meta::type_of(mem) :]>) {
+        return tag_invoke(deserialize_tag{}, val, out.[:mem:]);
+      } else {
+        return INCORRECT_TYPE;
+      }
+    } else {
+      return key_selector_reflection_detail::deserialize_member<mem>(val, out.[:mem:]);
+    }
+  } else {
+  static_assert(key_selector_reflection_detail::accepted_keys_are_distinct<T>(),
+                "two members of this structure accept the same JSON key (check rename, alias, "
+                "rename_all and flatten)");
   rvv_vls::ondemand::object obj;
   if constexpr (std::is_same_v<std::remove_cvref_t<ValT>, rvv_vls::ondemand::object>) {
     obj = val;
   } else {
     SIMDJSON_TRY(val.get_object().get(obj));
   }
-  template for (constexpr auto mem : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
-    if constexpr (!std::meta::is_const(mem) && std::meta::is_public(mem)) {
-      constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
-      if constexpr (concepts::optional_type<decltype(out.[:mem:])>) {
-        // for optional members, it's ok if the key is missing
-        auto error = obj[key].get(out.[:mem:]);
-        if (error && error != NO_SUCH_FIELD) {
-          if(error == NO_SUCH_FIELD) {
-            out.[:mem:].reset();
-            continue;
-          }
-          return error;
-        }
-      } else {
-        // for non-optional members, the key must be present
-        SIMDJSON_TRY(obj[key].get(out.[:mem:]));
+  if constexpr (simdjson::detail::has_annotation(^^T, ^^simdjson::detail::deny_unknown_fields_tag)) {
+    return key_selector_reflection_detail::deserialize_struct_scan<true>(obj, out);
+  } else {
+#if defined(SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION) && SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION
+  // Opt-out build: use the ordered per-member path, unless obj[key] cannot
+  // match T's keys.
+  if constexpr (key_selector_reflection_detail::keys_need_unescaping<T>()) {
+    return key_selector_reflection_detail::deserialize_struct_scan<false>(obj, out);
+  } else {
+    return key_selector_reflection_detail::deserialize_struct_ordered(obj, out);
+  }
+#else
+  if constexpr (key_selector_reflection_detail::eligible_field_count<T>() == 0) {
+    // No fields to deserialize: an empty key_selector cannot be built, so just
+    // validate that the input is an object (done above) and succeed. Mirrors the
+    // ordered per-member path, which iterates over zero members.
+    (void)out;
+    (void)obj;
+    return SUCCESS;
+  } else if constexpr (!key_selector_reflection_detail::keys_fit_selector<T>()) {
+    // Automatic fallback: T's accepted keys do not fit the key_selector limits
+    // (e.g. a member name longer than 63 characters, or a key with a double
+    // quote), so building a selector would be a compile error. Scan the object
+    // instead, so the default never breaks a struct that the opt-out path would
+    // accept.
+    return key_selector_reflection_detail::deserialize_struct_scan<false>(obj, out);
+  } else {
+  using selector = key_selector_reflection_detail::selector_for<T>;
+  if constexpr (key_selector_reflection_detail::all_eligible_fields_required<T>()
+                && !key_selector_reflection_detail::has_aliases<T>()) {
+    // Fast path: every member is required and has a single key. A single
+    // for_each pass parses each matched field; the returned match count then
+    // tells us whether every member was present (matched_count ==
+    // selector::size()) without a per-member "seen" array. A value-parse error
+    // (e.g. a type mismatch) is propagated by for_each.
+    auto walk = obj.template for_each<selector>(
+        [&](std::size_t matched_index, rvv_vls::ondemand::value field_value) -> error_code {
+      std::size_t counter = 0;
+      template for (constexpr auto path : std::define_static_array(key_selector_reflection_detail::eligible_fields(^^T))) {
+        using field = [: path :];
+        if (matched_index == counter) { return key_selector_reflection_detail::deserialize_member<field::leaf>(field_value, field::get(out)); }
+        ++counter;
       }
-    }
-  };
-  return simdjson::SUCCESS;
-}
-
-// Support for enum deserialization - deserialize from string representation using expand approach from P2996R12
+      return SUCCESS;
+    });
+    if (walk.error) { return walk.error; }
+    // A missing required member shows up as a short match count and is reported as
+    // NO_SUCH_FIELD, mirroring the ordered obj[key] path.
+    if (walk.matched_count != selector::size()) { return NO_SUCH_FIELD; }
+    return SUCCESS;
+  } else {
+    static constexpr auto key_fields = std::define_static_array(key_selector_reflection_detail::accepted_key_fields(^^T));
+    std::array<bool, key_selector_reflection_detail::eligible_field_count<T>()> seen_field{};
+    // Single pass over the object: each field whose key matches a member (or one
+    // of its aliases) yields its selector index, which we map back to the
+    // corresponding member. The first key seen for a member wins. The callback
+    // returns an error_code so that a value-parse error (e.g. a type mismatch on
+    // a matched field) is propagated by for_each instead of being silently dropped.
+    error_code walk_error = obj.template for_each<selector>(
+        [&](std::size_t matched_index, rvv_vls::ondemand::value field_value) -> error_code {
+      const std::size_t field_index = key_fields[matched_index];
+      if (seen_field[field_index]) { return SUCCESS; }
+      seen_field[field_index] = true;
+      return key_selector_reflection_detail::deserialize_field_at(field_index, field_value, out);
+    });
+    if (walk_error) { return walk_error; }
+    // Required members must be present: a missing one is reported as
+    // NO_SUCH_FIELD, mirroring the ordered obj[key] path. Optional and defaulted
+    // members may be absent.
+    return key_selector_reflection_detail::handle_missing_fields(seen_field, out);
+  }
+  }
+#endif // SIMDJSON_DISABLE_KEY_SELECTOR_REFLECTION
+  }
+  }
+}
+
+// Support for enum deserialization - deserialize from string representation using
+// expand approach from P2996R12. The accepted strings are the enumerator's JSON key
+// (see rename and rename_all) and its aliases.
 template <typename T, typename ValT>
   requires(std::is_enum_v<T>)
 error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
 #if SIMDJSON_STATIC_REFLECTION
   std::string_view str;
   SIMDJSON_TRY(val.get_string().get(str));
-  constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
+  static constexpr auto enumerators = std::define_static_array(std::meta::enumerators_of(^^T));
   template for (constexpr auto enum_val : enumerators) {
-    if (str == std::meta::identifier_of(enum_val)) {
-      out = [:enum_val:];
-      return SUCCESS;
+    template for (constexpr const char *key : std::define_static_array(key_selector_reflection_detail::accepted_keys_of(enum_val))) {
+      if (str == std::string_view(key)) {
+        out = [:enum_val:];
+        return SUCCESS;
+      }
     }
   };

@@ -179157,33 +230731,25 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {

 template <typename simdjson_value, typename T>
   requires(user_defined_type<std::remove_cvref_t<T>>)
-error_code tag_invoke(deserialize_tag, simdjson_value &val, std::unique_ptr<T> &out) noexcept {
-  if (!out) {
-    out = std::make_unique<T>();
-    if (!out) {
-      return MEMALLOC;
-    }
-  }
-  if (auto err = val.get(*out)) {
-    out.reset();
-    return err;
+error_code tag_invoke(deserialize_tag, simdjson_value &val, std::unique_ptr<T> &out) noexcept(false) {
+  std::unique_ptr<T> ptr(new (std::nothrow) T());
+  if (!ptr) {
+    return MEMALLOC;
   }
+  SIMDJSON_TRY(val.get(*ptr));
+  out = std::move(ptr);
   return SUCCESS;
 }

 template <typename simdjson_value, typename T>
   requires(user_defined_type<std::remove_cvref_t<T>>)
-error_code tag_invoke(deserialize_tag, simdjson_value &val, std::shared_ptr<T> &out) noexcept {
-  if (!out) {
-    out = std::make_shared<T>();
-    if (!out) {
-      return MEMALLOC;
-    }
-  }
-  if (auto err = val.get(*out)) {
-    out.reset();
-    return err;
+error_code tag_invoke(deserialize_tag, simdjson_value &val, std::shared_ptr<T> &out) noexcept(false) {
+  std::shared_ptr<T> ptr(new (std::nothrow) T());
+  if (!ptr) {
+    return MEMALLOC;
   }
+  SIMDJSON_TRY(val.get(*ptr));
+  out = std::move(ptr);
   return SUCCESS;
 }

@@ -179495,9 +231061,17 @@ simdjson_inline simdjson_result<array> array::started(value_iterator &iter) noex
   return array(iter);
 }

-simdjson_inline simdjson_result<array_iterator> array::begin() noexcept {
+simdjson_inline simdjson_result<array_iterator> array::begin() & noexcept {
 #if SIMDJSON_DEVELOPMENT_CHECKS
-  if (!iter.is_at_iterator_start()) { return OUT_OF_ORDER_ITERATION; }
+  if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
+  return array_iterator(iter, this);
+#endif
+  return array_iterator(iter);
+}
+simdjson_inline simdjson_result<array_iterator> array::begin() && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  // The array is a temporary that the iterator may outlive: do not lock it.
+  if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
 #endif
   return array_iterator(iter);
 }
@@ -179524,6 +231098,9 @@ simdjson_inline simdjson_result<std::string_view> array::raw_json() noexcept {
 SIMDJSON_PUSH_DISABLE_WARNINGS
 SIMDJSON_DISABLE_STRICT_OVERFLOW_WARNING
 simdjson_inline simdjson_result<size_t> array::count_elements() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   size_t count{0};
   // Important: we do not consume any of the values.
   for(simdjson_unused auto v : *this) { count++; }
@@ -179537,6 +231114,9 @@ simdjson_inline simdjson_result<size_t> array::count_elements() & noexcept {
 SIMDJSON_POP_DISABLE_WARNINGS

 simdjson_inline simdjson_result<bool> array::is_empty() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   bool is_not_empty;
   auto error = iter.reset_array().get(is_not_empty);
   if(error) { return error; }
@@ -179544,31 +231124,30 @@ simdjson_inline simdjson_result<bool> array::is_empty() & noexcept {
 }

 inline simdjson_result<bool> array::reset() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   return iter.reset_array();
 }

+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline void array::set_locked(bool _locked) noexcept {
+  locked = _locked;
+}
+#endif
+
 inline simdjson_result<value> array::at_pointer(std::string_view json_pointer) noexcept {
-  if (json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
+  // An empty pointer has no json_pointer[0]: with a default-constructed
+  // std::string_view, reading it dereferences a null pointer.
+  if (json_pointer.empty() || json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
   json_pointer = json_pointer.substr(1);
   // - means "the append position" or "the element after the end of the array"
   // We don't support this, because we're returning a real element, not a position.
   if (json_pointer == "-") { return INDEX_OUT_OF_BOUNDS; }

-  // Read the array index
   size_t array_index = 0;
   size_t i;
-  for (i = 0; i < json_pointer.length() && json_pointer[i] != '/'; i++) {
-    uint8_t digit = uint8_t(json_pointer[i] - '0');
-    // Check for non-digit in array index. If it's there, we're trying to get a field in an object
-    if (digit > 9) { return INCORRECT_TYPE; }
-    array_index = array_index*10 + digit;
-  }
-
-  // 0 followed by other digits is invalid
-  if (i > 1 && json_pointer[0] == '0') { return INVALID_JSON_POINTER; } // "JSON pointer array index has other characters after 0"
-
-  // Empty string is invalid; so is a "/" with no digits before it
-  if (i == 0) { return INVALID_JSON_POINTER; } // "Empty string in JSON pointer array index"
+  SIMDJSON_TRY(internal::parse_json_pointer_array_index(json_pointer, array_index, i));
   // Get the child
   auto child = at(array_index);
   // If there is an error, it ends here
@@ -179642,6 +231221,9 @@ inline error_code array::for_each_at_path_with_wildcard(std::string_view json_pa
 }

 simdjson_inline simdjson_result<value> array::at(size_t index) noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   size_t i = 0;
   for (auto value : *this) {
     if (i == index) { return value; }
@@ -179671,10 +231253,14 @@ simdjson_inline simdjson_result<rvv_vls::ondemand::array>::simdjson_result(
 {
 }

-simdjson_inline simdjson_result<rvv_vls::ondemand::array_iterator> simdjson_result<rvv_vls::ondemand::array>::begin() noexcept {
+simdjson_inline simdjson_result<rvv_vls::ondemand::array_iterator> simdjson_result<rvv_vls::ondemand::array>::begin() & noexcept {
   if (error()) { return error(); }
   return first.begin();
 }
+simdjson_inline simdjson_result<rvv_vls::ondemand::array_iterator> simdjson_result<rvv_vls::ondemand::array>::begin() && noexcept {
+  if (error()) { return error(); }
+  return std::move(first).begin();
+}
 simdjson_inline simdjson_result<rvv_vls::ondemand::array_iterator> simdjson_result<rvv_vls::ondemand::array>::end() noexcept {
   if (error()) { return error(); }
   return first.end();
@@ -179737,6 +231323,59 @@ simdjson_inline array_iterator::array_iterator(const value_iterator &_iter) noex
   : iter{_iter}
 {}

+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline array_iterator::array_iterator(const value_iterator &_iter, array* _parent) noexcept
+  : parent{_parent}, iter{_iter}
+{
+  if (parent) parent->set_locked(true);
+}
+
+simdjson_inline array_iterator::~array_iterator() noexcept
+{
+  if (parent) parent->set_locked(false);
+}
+
+simdjson_inline array_iterator::array_iterator(array_iterator&& other) noexcept
+  : has_been_referenced{other.has_been_referenced},
+    parent{other.parent},
+    iter{std::move(other.iter)}
+{
+  other.parent = nullptr;
+}
+
+simdjson_inline array_iterator& array_iterator::operator=(array_iterator&& other) noexcept {
+  if (this != &other)
+  {
+    if (parent)
+      parent->set_locked(false);
+    has_been_referenced = other.has_been_referenced;
+    parent = other.parent;
+    iter = std::move(other.iter);
+
+    other.parent = nullptr;
+  }
+  return *this;
+}
+
+simdjson_inline array_iterator::array_iterator(const array_iterator& other) noexcept
+  : has_been_referenced{other.has_been_referenced},
+    parent{nullptr},
+    iter{other.iter}
+{}
+
+simdjson_inline array_iterator& array_iterator::operator=(const array_iterator& other) noexcept {
+  if (this != &other)
+  {
+    if (parent)
+      parent->set_locked(false);
+    has_been_referenced = other.has_been_referenced;
+    parent = nullptr;
+    iter = other.iter;
+  }
+  return *this;
+}
+#endif
+
 simdjson_inline simdjson_result<value> array_iterator::operator*() noexcept {
 #if SIMDJSON_DEVELOPMENT_CHECKS
    SIMDJSON_ASSUME(!has_been_referenced);
@@ -179832,6 +231471,41 @@ namespace simdjson {
 namespace rvv_vls {
 namespace ondemand {

+/**
+ * @private Narrows the result of get_uint64() to a smaller unsigned integer type T.
+ * @returns NUMBER_OUT_OF_RANGE if the value does not fit in a T.
+ */
+template <typename T>
+simdjson_inline simdjson_result<T> narrow_integer(simdjson_result<uint64_t> wide) noexcept {
+  uint64_t result;
+  SIMDJSON_TRY(std::move(wide).get(result));
+  if (result > static_cast<uint64_t>((std::numeric_limits<T>::max)())) { return NUMBER_OUT_OF_RANGE; }
+  return static_cast<T>(result);
+}
+
+/**
+ * @private Narrows the result of get_int64() to a smaller signed integer type T.
+ * @returns NUMBER_OUT_OF_RANGE if the value does not fit in a T.
+ */
+template <typename T>
+simdjson_inline simdjson_result<T> narrow_integer(simdjson_result<int64_t> wide) noexcept {
+  int64_t result;
+  SIMDJSON_TRY(std::move(wide).get(result));
+  if (result > static_cast<int64_t>((std::numeric_limits<T>::max)()) || result < static_cast<int64_t>((std::numeric_limits<T>::min)())) { return NUMBER_OUT_OF_RANGE; }
+  return static_cast<T>(result);
+}
+
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+// get_float32() parses with get_float(), so float must be binary32.
+static_assert(std::numeric_limits<float>::is_iec559 && std::numeric_limits<float>::digits == 24,
+              "get_float32() requires float to be IEEE binary32");
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+// get_float64() parses with get_double(), so double must be binary64.
+static_assert(std::numeric_limits<double>::is_iec559 && std::numeric_limits<double>::digits == 53,
+              "get_float64() requires double to be IEEE binary64");
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
+
 simdjson_inline value::value(const value_iterator &_iter) noexcept
   : iter{_iter}
 {
@@ -179863,6 +231537,13 @@ simdjson_inline simdjson_result<raw_json_string> value::get_raw_json_string() no
 simdjson_inline simdjson_result<std::string_view> value::get_string(bool allow_replacement) noexcept {
   return iter.get_string(allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> value::get_u8string(bool allow_replacement) noexcept {
+  std::string_view content;
+  SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+  return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code value::get_string(string_type& receiver, bool allow_replacement) noexcept {
   return iter.get_string(receiver, allow_replacement);
@@ -179876,6 +231557,12 @@ simdjson_inline simdjson_result<double> value::get_double() noexcept {
 simdjson_inline simdjson_result<double> value::get_double_in_string() noexcept {
   return iter.get_double_in_string();
 }
+simdjson_inline simdjson_result<float> value::get_float() noexcept {
+  return iter.get_float();
+}
+simdjson_inline simdjson_result<float> value::get_float_in_string() noexcept {
+  return iter.get_float_in_string();
+}
 simdjson_inline simdjson_result<uint64_t> value::get_uint64() noexcept {
   return iter.get_uint64();
 }
@@ -179889,17 +231576,37 @@ simdjson_inline simdjson_result<int64_t> value::get_int64_in_string() noexcept {
   return iter.get_int64_in_string();
 }
 simdjson_inline simdjson_result<uint32_t> value::get_uint32() noexcept {
-  uint64_t result;
-  SIMDJSON_TRY(get_uint64().get(result));
-  if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<uint32_t>(result);
+  return narrow_integer<uint32_t>(get_uint64());
 }
 simdjson_inline simdjson_result<int32_t> value::get_int32() noexcept {
-  int64_t result;
-  SIMDJSON_TRY(get_int64().get(result));
-  if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<int32_t>(result);
+  return narrow_integer<int32_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint16_t> value::get_uint16() noexcept {
+  return narrow_integer<uint16_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int16_t> value::get_int16() noexcept {
+  return narrow_integer<int16_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint8_t> value::get_uint8() noexcept {
+  return narrow_integer<uint8_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int8_t> value::get_int8() noexcept {
+  return narrow_integer<int8_t>(get_int64());
 }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> value::get_float32() noexcept {
+  float result;
+  SIMDJSON_TRY(get_float().get(result));
+  return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> value::get_float64() noexcept {
+  double result;
+  SIMDJSON_TRY(get_double().get(result));
+  return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<bool> value::get_bool() noexcept {
   return iter.get_bool();
 }
@@ -179911,12 +231618,26 @@ template<> simdjson_inline simdjson_result<array> value::get() noexcept { return
 template<> simdjson_inline simdjson_result<object> value::get() noexcept { return get_object(); }
 template<> simdjson_inline simdjson_result<raw_json_string> value::get() noexcept { return get_raw_json_string(); }
 template<> simdjson_inline simdjson_result<std::string_view> value::get() noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> value::get() noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_inline simdjson_result<number> value::get() noexcept { return get_number(); }
 template<> simdjson_inline simdjson_result<double> value::get() noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> value::get() noexcept { return get_float(); }
 template<> simdjson_inline simdjson_result<uint64_t> value::get() noexcept { return get_uint64(); }
 template<> simdjson_inline simdjson_result<int64_t> value::get() noexcept { return get_int64(); }
 template<> simdjson_inline simdjson_result<uint32_t> value::get() noexcept { return get_uint32(); }
 template<> simdjson_inline simdjson_result<int32_t> value::get() noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> value::get() noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> value::get() noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> value::get() noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> value::get() noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> value::get() noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> value::get() noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<> simdjson_inline simdjson_result<bool> value::get() noexcept { return get_bool(); }


@@ -179924,12 +231645,26 @@ template<> simdjson_warn_unused simdjson_inline error_code value::get(array& out
 template<> simdjson_warn_unused simdjson_inline error_code value::get(object& out) noexcept { return get_object().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(raw_json_string& out) noexcept { return get_raw_json_string().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(std::string_view& out) noexcept { return get_string(false).get(out); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::u8string_view& out) noexcept { return get_u8string(false).get(out); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_warn_unused simdjson_inline error_code value::get(number& out) noexcept { return get_number().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(double& out) noexcept { return get_double().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(float& out) noexcept { return get_float().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(uint64_t& out) noexcept { return get_uint64().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(int64_t& out) noexcept { return get_int64().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(uint32_t& out) noexcept { return get_uint32().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code value::get(int32_t& out) noexcept { return get_int32().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(uint16_t& out) noexcept { return get_uint16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(int16_t& out) noexcept { return get_int16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(uint8_t& out) noexcept { return get_uint8().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code value::get(int8_t& out) noexcept { return get_int8().get(out); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::float32_t& out) noexcept { return get_float32().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_warn_unused simdjson_inline error_code value::get(std::float64_t& out) noexcept { return get_float64().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<>  simdjson_warn_unused simdjson_inline error_code value::get(bool& out) noexcept { return get_bool().get(out); }

 #if SIMDJSON_EXCEPTIONS
@@ -180098,6 +231833,9 @@ inline bool is_pointer_well_formed(std::string_view json_pointer) noexcept {
 }

 simdjson_inline simdjson_result<value> value::at_pointer(std::string_view json_pointer) noexcept {
+  // The empty JSON Pointer refers to the whole value (RFC 6901), as in
+  // document::at_pointer.
+  if (json_pointer.empty()) { return value(iter); }
   json_type t;
   SIMDJSON_TRY(type().get(t));
   switch (t)
@@ -180135,6 +231873,10 @@ template <typename Func>
 template <typename Func>
 #endif
 inline error_code value::for_each_at_path_with_wildcard(std::string_view json_path, Func&& callback) noexcept {
+  // Every recursive step of for_each_at_path_with_wildcard goes through this
+  // function, and each one descends one level into the document. A path with
+  // many segments applied to a deeply nested document would otherwise recurse
+  // without bound and overflow the stack.
   if (size_t(iter.depth()) >= iter.json_iter().parser->max_depth()) { return DEPTH_ERROR; }
   json_type t;
   SIMDJSON_TRY(type().get(t));
@@ -180248,10 +231990,46 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<rvv_vls::ondemand::valu
   if (error()) { return error(); }
   return first.get_int32();
 }
+simdjson_inline simdjson_result<uint16_t> simdjson_result<rvv_vls::ondemand::value>::get_uint16() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<rvv_vls::ondemand::value>::get_int16() noexcept {
+  if (error()) { return error(); }
+  return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<rvv_vls::ondemand::value>::get_uint8() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<rvv_vls::ondemand::value>::get_int8() noexcept {
+  if (error()) { return error(); }
+  return first.get_int8();
+}
 simdjson_inline simdjson_result<double> simdjson_result<rvv_vls::ondemand::value>::get_double() noexcept {
   if (error()) { return error(); }
   return first.get_double();
 }
+simdjson_inline simdjson_result<float> simdjson_result<rvv_vls::ondemand::value>::get_float() noexcept {
+  if (error()) { return error(); }
+  return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<rvv_vls::ondemand::value>::get_float_in_string() noexcept {
+  if (error()) { return error(); }
+  return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<rvv_vls::ondemand::value>::get_float32() noexcept {
+  if (error()) { return error(); }
+  return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<rvv_vls::ondemand::value>::get_float64() noexcept {
+  if (error()) { return error(); }
+  return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<double> simdjson_result<rvv_vls::ondemand::value>::get_double_in_string() noexcept {
   if (error()) { return error(); }
   return first.get_double_in_string();
@@ -180260,6 +232038,12 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<rvv_vls::ondem
   if (error()) { return error(); }
   return first.get_string(allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<rvv_vls::ondemand::value>::get_u8string(bool allow_replacement) noexcept {
+  if (error()) { return error(); }
+  return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_inline error_code simdjson_result<rvv_vls::ondemand::value>::get_string(string_type& receiver, bool allow_replacement) noexcept {
   if (error()) { return error(); }
@@ -180288,11 +232072,23 @@ template<> simdjson_inline error_code simdjson_result<rvv_vls::ondemand::value>:
   return SUCCESS;
 }

-template<typename T> simdjson_inline simdjson_result<T> simdjson_result<rvv_vls::ondemand::value>::get() noexcept {
+template<typename T> simdjson_inline simdjson_result<T> simdjson_result<rvv_vls::ondemand::value>::get()
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, rvv_vls::ondemand::value>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>();
 }
-template<typename T> simdjson_inline error_code simdjson_result<rvv_vls::ondemand::value>::get(T &out) noexcept {
+template<typename T> simdjson_inline error_code simdjson_result<rvv_vls::ondemand::value>::get(T &out)
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, rvv_vls::ondemand::value>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>(out);
 }
@@ -180562,16 +232358,22 @@ simdjson_inline simdjson_result<int64_t> document::get_int64_in_string() noexcep
   return get_root_value_iterator().get_root_int64_in_string(true);
 }
 simdjson_inline simdjson_result<uint32_t> document::get_uint32() noexcept {
-  uint64_t result;
-  SIMDJSON_TRY(get_uint64().get(result));
-  if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<uint32_t>(result);
+  return narrow_integer<uint32_t>(get_uint64());
 }
 simdjson_inline simdjson_result<int32_t> document::get_int32() noexcept {
-  int64_t result;
-  SIMDJSON_TRY(get_int64().get(result));
-  if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<int32_t>(result);
+  return narrow_integer<int32_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint16_t> document::get_uint16() noexcept {
+  return narrow_integer<uint16_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int16_t> document::get_int16() noexcept {
+  return narrow_integer<int16_t>(get_int64());
+}
+simdjson_inline simdjson_result<uint8_t> document::get_uint8() noexcept {
+  return narrow_integer<uint8_t>(get_uint64());
+}
+simdjson_inline simdjson_result<int8_t> document::get_int8() noexcept {
+  return narrow_integer<int8_t>(get_int64());
 }
 simdjson_inline simdjson_result<double> document::get_double() noexcept {
   return get_root_value_iterator().get_root_double(true);
@@ -180579,9 +232381,36 @@ simdjson_inline simdjson_result<double> document::get_double() noexcept {
 simdjson_inline simdjson_result<double> document::get_double_in_string() noexcept {
   return get_root_value_iterator().get_root_double_in_string(true);
 }
+simdjson_inline simdjson_result<float> document::get_float() noexcept {
+  return get_root_value_iterator().get_root_float(true);
+}
+simdjson_inline simdjson_result<float> document::get_float_in_string() noexcept {
+  return get_root_value_iterator().get_root_float_in_string(true);
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> document::get_float32() noexcept {
+  float result;
+  SIMDJSON_TRY(get_float().get(result));
+  return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> document::get_float64() noexcept {
+  double result;
+  SIMDJSON_TRY(get_double().get(result));
+  return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<std::string_view> document::get_string(bool allow_replacement) noexcept {
   return get_root_value_iterator().get_root_string(true, allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> document::get_u8string(bool allow_replacement) noexcept {
+  std::string_view content;
+  SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+  return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code document::get_string(string_type& receiver, bool allow_replacement) noexcept {
   return get_root_value_iterator().get_root_string(receiver, true, allow_replacement);
@@ -180603,11 +232432,25 @@ template<> simdjson_inline simdjson_result<array> document::get() & noexcept { r
 template<> simdjson_inline simdjson_result<object> document::get() & noexcept { return get_object(); }
 template<> simdjson_inline simdjson_result<raw_json_string> document::get() & noexcept { return get_raw_json_string(); }
 template<> simdjson_inline simdjson_result<std::string_view> document::get() & noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> document::get() & noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_inline simdjson_result<double> document::get() & noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> document::get() & noexcept { return get_float(); }
 template<> simdjson_inline simdjson_result<uint64_t> document::get() & noexcept { return get_uint64(); }
 template<> simdjson_inline simdjson_result<int64_t> document::get() & noexcept { return get_int64(); }
 template<> simdjson_inline simdjson_result<uint32_t> document::get() & noexcept { return get_uint32(); }
 template<> simdjson_inline simdjson_result<int32_t> document::get() & noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> document::get() & noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> document::get() & noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> document::get() & noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> document::get() & noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> document::get() & noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> document::get() & noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<> simdjson_inline simdjson_result<bool> document::get() & noexcept { return get_bool(); }
 template<> simdjson_inline simdjson_result<value> document::get() & noexcept { return get_value(); }

@@ -180615,17 +232458,35 @@ template<> simdjson_warn_unused simdjson_inline error_code document::get(array&
 template<> simdjson_warn_unused simdjson_inline error_code document::get(object& out) & noexcept { return get_object().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(raw_json_string& out) & noexcept { return get_raw_json_string().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(std::string_view& out) & noexcept { return get_string(false).get(out); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::u8string_view& out) & noexcept { return get_u8string(false).get(out); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_warn_unused simdjson_inline error_code document::get(double& out) & noexcept { return get_double().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(float& out) & noexcept { return get_float().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(uint64_t& out) & noexcept { return get_uint64().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(int64_t& out) & noexcept { return get_int64().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(uint32_t& out) & noexcept { return get_uint32().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(int32_t& out) & noexcept { return get_int32().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(uint16_t& out) & noexcept { return get_uint16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(int16_t& out) & noexcept { return get_int16().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(uint8_t& out) & noexcept { return get_uint8().get(out); }
+template<> simdjson_warn_unused simdjson_inline error_code document::get(int8_t& out) & noexcept { return get_int8().get(out); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::float32_t& out) & noexcept { return get_float32().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_warn_unused simdjson_inline error_code document::get(std::float64_t& out) & noexcept { return get_float64().get(out); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<> simdjson_warn_unused simdjson_inline error_code document::get(bool& out) & noexcept { return get_bool().get(out); }
 template<> simdjson_warn_unused simdjson_inline error_code document::get(value& out) & noexcept { return get_value().get(out); }

 template<> simdjson_deprecated simdjson_inline simdjson_result<raw_json_string> document::get() && noexcept { return get_raw_json_string(); }
 template<> simdjson_deprecated simdjson_inline simdjson_result<std::string_view> document::get() && noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_deprecated simdjson_inline simdjson_result<std::u8string_view> document::get() && noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_deprecated simdjson_inline simdjson_result<double> document::get() && noexcept { return std::forward<document>(*this).get_double(); }
+template<> simdjson_deprecated simdjson_inline simdjson_result<float> document::get() && noexcept { return std::forward<document>(*this).get_float(); }
 template<> simdjson_deprecated simdjson_inline simdjson_result<uint64_t> document::get() && noexcept { return std::forward<document>(*this).get_uint64(); }
 template<> simdjson_deprecated simdjson_inline simdjson_result<int64_t> document::get() && noexcept { return std::forward<document>(*this).get_int64(); }
 template<> simdjson_deprecated simdjson_inline simdjson_result<bool> document::get() && noexcept { return std::forward<document>(*this).get_bool(); }
@@ -180964,6 +232825,22 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<rvv_vls::ondemand::docu
   if (error()) { return error(); }
   return first.get_int32();
 }
+simdjson_inline simdjson_result<uint16_t> simdjson_result<rvv_vls::ondemand::document>::get_uint16() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<rvv_vls::ondemand::document>::get_int16() noexcept {
+  if (error()) { return error(); }
+  return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<rvv_vls::ondemand::document>::get_uint8() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<rvv_vls::ondemand::document>::get_int8() noexcept {
+  if (error()) { return error(); }
+  return first.get_int8();
+}
 simdjson_inline simdjson_result<double> simdjson_result<rvv_vls::ondemand::document>::get_double() noexcept {
   if (error()) { return error(); }
   return first.get_double();
@@ -180972,10 +232849,36 @@ simdjson_inline simdjson_result<double> simdjson_result<rvv_vls::ondemand::docum
   if (error()) { return error(); }
   return first.get_double_in_string();
 }
+simdjson_inline simdjson_result<float> simdjson_result<rvv_vls::ondemand::document>::get_float() noexcept {
+  if (error()) { return error(); }
+  return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<rvv_vls::ondemand::document>::get_float_in_string() noexcept {
+  if (error()) { return error(); }
+  return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<rvv_vls::ondemand::document>::get_float32() noexcept {
+  if (error()) { return error(); }
+  return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<rvv_vls::ondemand::document>::get_float64() noexcept {
+  if (error()) { return error(); }
+  return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<std::string_view> simdjson_result<rvv_vls::ondemand::document>::get_string(bool allow_replacement) noexcept {
   if (error()) { return error(); }
   return first.get_string(allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<rvv_vls::ondemand::document>::get_u8string(bool allow_replacement) noexcept {
+  if (error()) { return error(); }
+  return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code simdjson_result<rvv_vls::ondemand::document>::get_string(string_type& receiver, bool allow_replacement) noexcept {
   if (error()) { return error(); }
@@ -181003,22 +232906,46 @@ simdjson_inline simdjson_result<bool> simdjson_result<rvv_vls::ondemand::documen
 }

 template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<rvv_vls::ondemand::document>::get() & noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<rvv_vls::ondemand::document>::get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, rvv_vls::ondemand::document>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>();
 }
 template<typename T>
-simdjson_deprecated simdjson_inline simdjson_result<T> simdjson_result<rvv_vls::ondemand::document>::get() && noexcept {
+simdjson_deprecated simdjson_inline simdjson_result<T> simdjson_result<rvv_vls::ondemand::document>::get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, rvv_vls::ondemand::document>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return std::forward<rvv_vls::ondemand::document>(first).get<T>();
 }
 template<typename T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<rvv_vls::ondemand::document>::get(T &out) & noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<rvv_vls::ondemand::document>::get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, rvv_vls::ondemand::document>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>(out);
 }
 template<typename T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<rvv_vls::ondemand::document>::get(T &out) && noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<rvv_vls::ondemand::document>::get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, rvv_vls::ondemand::document>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return std::forward<rvv_vls::ondemand::document>(first).get<T>(out);
 }
@@ -181087,27 +233014,27 @@ simdjson_inline simdjson_result<rvv_vls::ondemand::document>::operator rvv_vls::
 }
 simdjson_inline simdjson_result<rvv_vls::ondemand::document>::operator uint64_t() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_uint64();
 }
 simdjson_inline simdjson_result<rvv_vls::ondemand::document>::operator int64_t() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_int64();
 }
 simdjson_inline simdjson_result<rvv_vls::ondemand::document>::operator double() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_double();
 }
 simdjson_inline simdjson_result<rvv_vls::ondemand::document>::operator std::string_view() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_string();
 }
 simdjson_inline simdjson_result<rvv_vls::ondemand::document>::operator rvv_vls::ondemand::raw_json_string() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_raw_json_string();
 }
 simdjson_inline simdjson_result<rvv_vls::ondemand::document>::operator bool() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_bool();
 }
 simdjson_inline simdjson_result<rvv_vls::ondemand::document>::operator rvv_vls::ondemand::value() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
@@ -181197,21 +233124,38 @@ simdjson_inline simdjson_result<uint64_t> document_reference::get_uint64() noexc
 simdjson_inline simdjson_result<uint64_t> document_reference::get_uint64_in_string() noexcept { return doc->get_root_value_iterator().get_root_uint64_in_string(false); }
 simdjson_inline simdjson_result<int64_t> document_reference::get_int64() noexcept { return doc->get_root_value_iterator().get_root_int64(false); }
 simdjson_inline simdjson_result<int64_t> document_reference::get_int64_in_string() noexcept { return doc->get_root_value_iterator().get_root_int64_in_string(false); }
-simdjson_inline simdjson_result<uint32_t> document_reference::get_uint32() noexcept {
-  uint64_t result;
-  SIMDJSON_TRY(doc->get_root_value_iterator().get_root_uint64(false).get(result));
-  if (result > (std::numeric_limits<uint32_t>::max)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<uint32_t>(result);
-}
-simdjson_inline simdjson_result<int32_t> document_reference::get_int32() noexcept {
-  int64_t result;
-  SIMDJSON_TRY(doc->get_root_value_iterator().get_root_int64(false).get(result));
-  if (result > (std::numeric_limits<int32_t>::max)() || result < (std::numeric_limits<int32_t>::min)()) { return NUMBER_OUT_OF_RANGE; }
-  return static_cast<int32_t>(result);
-}
+simdjson_inline simdjson_result<uint32_t> document_reference::get_uint32() noexcept { return narrow_integer<uint32_t>(get_uint64()); }
+simdjson_inline simdjson_result<int32_t> document_reference::get_int32() noexcept { return narrow_integer<int32_t>(get_int64()); }
+simdjson_inline simdjson_result<uint16_t> document_reference::get_uint16() noexcept { return narrow_integer<uint16_t>(get_uint64()); }
+simdjson_inline simdjson_result<int16_t> document_reference::get_int16() noexcept { return narrow_integer<int16_t>(get_int64()); }
+simdjson_inline simdjson_result<uint8_t> document_reference::get_uint8() noexcept { return narrow_integer<uint8_t>(get_uint64()); }
+simdjson_inline simdjson_result<int8_t> document_reference::get_int8() noexcept { return narrow_integer<int8_t>(get_int64()); }
 simdjson_inline simdjson_result<double> document_reference::get_double() noexcept { return doc->get_root_value_iterator().get_root_double(false); }
-simdjson_inline simdjson_result<double> document_reference::get_double_in_string() noexcept { return doc->get_root_value_iterator().get_root_double(false); }
+simdjson_inline simdjson_result<double> document_reference::get_double_in_string() noexcept { return doc->get_root_value_iterator().get_root_double_in_string(false); }
+simdjson_inline simdjson_result<float> document_reference::get_float() noexcept { return doc->get_root_value_iterator().get_root_float(false); }
+simdjson_inline simdjson_result<float> document_reference::get_float_in_string() noexcept { return doc->get_root_value_iterator().get_root_float_in_string(false); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> document_reference::get_float32() noexcept {
+  float result;
+  SIMDJSON_TRY(get_float().get(result));
+  return static_cast<std::float32_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> document_reference::get_float64() noexcept {
+  double result;
+  SIMDJSON_TRY(get_double().get(result));
+  return static_cast<std::float64_t>(result);
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<std::string_view> document_reference::get_string(bool allow_replacement) noexcept { return doc->get_root_value_iterator().get_root_string(false, allow_replacement); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> document_reference::get_u8string(bool allow_replacement) noexcept {
+  std::string_view content;
+  SIMDJSON_TRY( get_string(allow_replacement).get(content) );
+  return internal::as_u8string_view(content);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code document_reference::get_string(string_type& receiver, bool allow_replacement) noexcept { return doc->get_root_value_iterator().get_root_string(receiver, false, allow_replacement); }
 simdjson_inline simdjson_result<std::string_view> document_reference::get_wobbly_string() noexcept { return doc->get_root_value_iterator().get_root_wobbly_string(false); }
@@ -181223,11 +233167,25 @@ template<> simdjson_inline simdjson_result<array> document_reference::get() & no
 template<> simdjson_inline simdjson_result<object> document_reference::get() & noexcept { return get_object(); }
 template<> simdjson_inline simdjson_result<raw_json_string> document_reference::get() & noexcept { return get_raw_json_string(); }
 template<> simdjson_inline simdjson_result<std::string_view> document_reference::get() & noexcept { return get_string(false); }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+template<> simdjson_inline simdjson_result<std::u8string_view> document_reference::get() & noexcept { return get_u8string(false); }
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template<> simdjson_inline simdjson_result<double> document_reference::get() & noexcept { return get_double(); }
+template<> simdjson_inline simdjson_result<float> document_reference::get() & noexcept { return get_float(); }
 template<> simdjson_inline simdjson_result<uint64_t> document_reference::get() & noexcept { return get_uint64(); }
 template<> simdjson_inline simdjson_result<int64_t> document_reference::get() & noexcept { return get_int64(); }
 template<> simdjson_inline simdjson_result<uint32_t> document_reference::get() & noexcept { return get_uint32(); }
 template<> simdjson_inline simdjson_result<int32_t> document_reference::get() & noexcept { return get_int32(); }
+template<> simdjson_inline simdjson_result<uint16_t> document_reference::get() & noexcept { return get_uint16(); }
+template<> simdjson_inline simdjson_result<int16_t> document_reference::get() & noexcept { return get_int16(); }
+template<> simdjson_inline simdjson_result<uint8_t> document_reference::get() & noexcept { return get_uint8(); }
+template<> simdjson_inline simdjson_result<int8_t> document_reference::get() & noexcept { return get_int8(); }
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+template<> simdjson_inline simdjson_result<std::float32_t> document_reference::get() & noexcept { return get_float32(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+template<> simdjson_inline simdjson_result<std::float64_t> document_reference::get() & noexcept { return get_float64(); }
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 template<> simdjson_inline simdjson_result<bool> document_reference::get() & noexcept { return get_bool(); }
 template<> simdjson_inline simdjson_result<value> document_reference::get() & noexcept { return get_value(); }
 #if SIMDJSON_EXCEPTIONS
@@ -181373,6 +233331,22 @@ simdjson_inline simdjson_result<int32_t> simdjson_result<rvv_vls::ondemand::docu
   if (error()) { return error(); }
   return first.get_int32();
 }
+simdjson_inline simdjson_result<uint16_t> simdjson_result<rvv_vls::ondemand::document_reference>::get_uint16() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint16();
+}
+simdjson_inline simdjson_result<int16_t> simdjson_result<rvv_vls::ondemand::document_reference>::get_int16() noexcept {
+  if (error()) { return error(); }
+  return first.get_int16();
+}
+simdjson_inline simdjson_result<uint8_t> simdjson_result<rvv_vls::ondemand::document_reference>::get_uint8() noexcept {
+  if (error()) { return error(); }
+  return first.get_uint8();
+}
+simdjson_inline simdjson_result<int8_t> simdjson_result<rvv_vls::ondemand::document_reference>::get_int8() noexcept {
+  if (error()) { return error(); }
+  return first.get_int8();
+}
 simdjson_inline simdjson_result<double> simdjson_result<rvv_vls::ondemand::document_reference>::get_double() noexcept {
   if (error()) { return error(); }
   return first.get_double();
@@ -181381,10 +233355,36 @@ simdjson_inline simdjson_result<double> simdjson_result<rvv_vls::ondemand::docum
   if (error()) { return error(); }
   return first.get_double_in_string();
 }
+simdjson_inline simdjson_result<float> simdjson_result<rvv_vls::ondemand::document_reference>::get_float() noexcept {
+  if (error()) { return error(); }
+  return first.get_float();
+}
+simdjson_inline simdjson_result<float> simdjson_result<rvv_vls::ondemand::document_reference>::get_float_in_string() noexcept {
+  if (error()) { return error(); }
+  return first.get_float_in_string();
+}
+#if SIMDJSON_SUPPORTS_FLOAT32_T
+simdjson_inline simdjson_result<std::float32_t> simdjson_result<rvv_vls::ondemand::document_reference>::get_float32() noexcept {
+  if (error()) { return error(); }
+  return first.get_float32();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT32_T
+#if SIMDJSON_SUPPORTS_FLOAT64_T
+simdjson_inline simdjson_result<std::float64_t> simdjson_result<rvv_vls::ondemand::document_reference>::get_float64() noexcept {
+  if (error()) { return error(); }
+  return first.get_float64();
+}
+#endif // SIMDJSON_SUPPORTS_FLOAT64_T
 simdjson_inline simdjson_result<std::string_view> simdjson_result<rvv_vls::ondemand::document_reference>::get_string(bool allow_replacement) noexcept {
   if (error()) { return error(); }
   return first.get_string(allow_replacement);
 }
+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<rvv_vls::ondemand::document_reference>::get_u8string(bool allow_replacement) noexcept {
+  if (error()) { return error(); }
+  return first.get_u8string(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
 template <typename string_type>
 simdjson_warn_unused simdjson_inline error_code simdjson_result<rvv_vls::ondemand::document_reference>::get_string(string_type& receiver, bool allow_replacement) noexcept {
   if (error()) { return error(); }
@@ -181411,22 +233411,46 @@ simdjson_inline simdjson_result<bool> simdjson_result<rvv_vls::ondemand::documen
   return first.is_null();
 }
 template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<rvv_vls::ondemand::document_reference>::get() & noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<rvv_vls::ondemand::document_reference>::get() &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, rvv_vls::ondemand::document_reference>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>();
 }
 template<typename T>
-simdjson_inline simdjson_result<T> simdjson_result<rvv_vls::ondemand::document_reference>::get() && noexcept {
+simdjson_inline simdjson_result<T> simdjson_result<rvv_vls::ondemand::document_reference>::get() &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, rvv_vls::ondemand::document_reference>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return std::forward<rvv_vls::ondemand::document_reference>(first).get<T>();
 }
 template <class T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<rvv_vls::ondemand::document_reference>::get(T &out) & noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<rvv_vls::ondemand::document_reference>::get(T &out) &
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, rvv_vls::ondemand::document_reference>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return first.get<T>(out);
 }
 template <class T>
-simdjson_warn_unused simdjson_inline error_code simdjson_result<rvv_vls::ondemand::document_reference>::get(T &out) && noexcept {
+simdjson_warn_unused simdjson_inline error_code simdjson_result<rvv_vls::ondemand::document_reference>::get(T &out) &&
+#if SIMDJSON_SUPPORTS_CONCEPTS
+    noexcept(nothrow_gettable<T, rvv_vls::ondemand::document_reference>)
+#else
+    noexcept
+#endif
+{
   if (error()) { return error(); }
   return std::forward<rvv_vls::ondemand::document_reference>(first).get<T>(out);
 }
@@ -181488,27 +233512,27 @@ simdjson_inline simdjson_result<rvv_vls::ondemand::document_reference>::operator
 }
 simdjson_inline simdjson_result<rvv_vls::ondemand::document_reference>::operator uint64_t() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_uint64();
 }
 simdjson_inline simdjson_result<rvv_vls::ondemand::document_reference>::operator int64_t() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_int64();
 }
 simdjson_inline simdjson_result<rvv_vls::ondemand::document_reference>::operator double() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_double();
 }
 simdjson_inline simdjson_result<rvv_vls::ondemand::document_reference>::operator std::string_view() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_string();
 }
 simdjson_inline simdjson_result<rvv_vls::ondemand::document_reference>::operator rvv_vls::ondemand::raw_json_string() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_raw_json_string();
 }
 simdjson_inline simdjson_result<rvv_vls::ondemand::document_reference>::operator bool() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
-  return first;
+  return first.get_bool();
 }
 simdjson_inline simdjson_result<rvv_vls::ondemand::document_reference>::operator rvv_vls::ondemand::value() noexcept(false) {
   if (error()) { throw simdjson_error(error()); }
@@ -181574,6 +233598,7 @@ simdjson_warn_unused simdjson_inline error_code simdjson_result<rvv_vls::ondeman
 /* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */

 #include <algorithm>
+#include <cstring>
 #include <stdexcept>

 namespace simdjson {
@@ -181660,23 +233685,20 @@ simdjson_inline document_stream::document_stream(
   const uint8_t *_buf,
   size_t _len,
   size_t _batch_size,
-  bool _allow_comma_separated
+  bool _allow_comma_separated,
+  stream_format _format
 ) noexcept
   : parser{&_parser},
     buf{_buf},
     len{_len},
     batch_size{_batch_size <= MINIMAL_BATCH_SIZE ? MINIMAL_BATCH_SIZE : _batch_size},
     allow_comma_separated{_allow_comma_separated},
+    format{_format},
     error{SUCCESS}
     #ifdef SIMDJSON_THREADS_ENABLED
     , use_thread(_parser.threaded) // we need to make a copy because _parser.threaded can change
     #endif
 {
-#ifdef SIMDJSON_THREADS_ENABLED
-  if(worker.get() == nullptr) {
-    error = MEMALLOC;
-  }
-#endif
 }

 simdjson_inline document_stream::document_stream() noexcept
@@ -181685,6 +233707,7 @@ simdjson_inline document_stream::document_stream() noexcept
     len{0},
     batch_size{0},
     allow_comma_separated{false},
+    format{stream_format::whitespace_delimited},
     error{UNINITIALIZED}
     #ifdef SIMDJSON_THREADS_ENABLED
     , use_thread(false)
@@ -181704,6 +233727,9 @@ inline size_t document_stream::size_in_bytes() const noexcept {
 }

 inline size_t document_stream::truncated_bytes() const noexcept {
+  // Stage 1 returns EMPTY on zero-length input before it writes the index
+  // sentinels read below, so they would still hold a previous stream's values.
+  if (len == 0) { return 0; }
   if(error == CAPACITY) { return len - batch_start; }
   return parser->implementation->structural_indexes[parser->implementation->n_structural_indexes] - parser->implementation->structural_indexes[parser->implementation->n_structural_indexes + 1];
 }
@@ -181784,13 +233810,20 @@ inline void document_stream::start() noexcept {
     error = run_stage1(*parser, batch_start);
   }
   if (error) { return; }
-  doc_index = batch_start;
+  // For json_sequence mode, structural_indexes[0] points to the actual JSON value
+  // after the RS delimiter and any following whitespace. For regular mode, it is
+  // the offset from batch_start to the first document in the batch.
+  doc_index = batch_start + parser->implementation->structural_indexes[0];
   doc = document(json_iterator(&buf[batch_start], parser));
   doc.iter._streaming = true;

   #ifdef SIMDJSON_THREADS_ENABLED
   if (use_thread && next_batch_start() < len) {
     // Kick off the first thread on next batch if needed
+    if (worker.get() == nullptr) {
+      worker.reset(new(std::nothrow) stage1_worker());
+      if (worker.get() == nullptr) { error = MEMALLOC; return; }
+    }
     error = stage1_thread_parser.allocate(batch_size);
     if (error) { return; }
     worker->start_thread();
@@ -181865,12 +233898,69 @@ inline void document_stream::next() noexcept {
        */

       if (error) { continue; } // If the error was EMPTY, we may want to load another batch.
-      doc_index = batch_start;
+      doc_index = batch_start + parser->implementation->structural_indexes[0];
     }
   }
 }

+simdjson_inline uint8_t document_stream::document_delimiter() const noexcept {
+  switch (format) {
+    case stream_format::newline_delimited: return '\n';
+    case stream_format::json_sequence: return 0x1E;
+    default: return 0;
+  }
+}
+
+simdjson_inline bool document_stream::skip_to_delimiter(uint8_t delimiter) noexcept {
+  const uint8_t *const base = &buf[batch_start];
+  const token_position pos = doc.iter.position();
+  const token_position end = doc.iter.end_position();
+  if (pos >= end) { return false; }
+  const size_t here = size_t(doc.iter.token.peek(pos) - base);
+  const size_t batch_len =
+      (len - batch_start < batch_size) ? len - batch_start : batch_size;
+  if (here >= batch_len) { return false; }
+  const uint8_t *const found = static_cast<const uint8_t *>(
+      std::memchr(base + here, delimiter, batch_len - here));
+  if (found == nullptr) { return false; }
+
+  const uint32_t boundary = uint32_t(found - base);
+  // The answer is near `pos`: the delimiter ends the current document, while
+  // `end` spans the whole batch. Gallop first so the cost follows the distance
+  // rather than the size of the batch.
+  token_position lo = pos;
+  size_t hop = 1;
+  while (lo + hop < end && lo[hop] < boundary) { lo += hop; hop <<= 1; }
+  token_position hi = (lo + hop < end) ? lo + hop : end;
+  while (lo < hi) {
+    const token_position mid = lo + ((hi - lo) >> 1);
+    if (*mid < boundary) { lo = mid + 1; } else { hi = mid; }
+  }
+  doc.iter.token.set_position(lo);
+  return true;
+}
+
 inline void document_stream::next_document() noexcept {
+  // A delimiter that cannot occur inside a document tells us where the current
+  // one ends, so we can jump there instead of walking every structural. Only
+  // valid while the iterator is still inside the document: a consumed document
+  // already sits on the next one's first token, and skip_child() returns at
+  // once for it.
+  //
+  // The jump does not structure-validate the unread remainder of the current
+  // document: under newline_delimited / json_sequence the next delimiter is
+  // assumed to be the true document boundary. Callers that leave depth() > 0
+  // while violating that contract (e.g. pretty multi-line JSON under
+  // newline_delimited) can mis-align following documents; use
+  // whitespace_delimited if unsure.
+  const uint8_t delimiter = document_delimiter();
+  if (delimiter != 0 && !error && doc.iter.depth() > 0 &&
+      skip_to_delimiter(delimiter)) {
+    doc.iter._depth = 1;
+    doc.iter._string_buf_loc = parser->string_buf.get();
+    doc.iter._root = doc.iter.position();
+    return;
+  }
   // Go to next place where depth=0 (document depth)
   error = doc.iter.skip_child(0);
   if (error) { return; }
@@ -181894,10 +233984,35 @@ inline error_code document_stream::run_stage1(ondemand::parser &p, size_t _batch
   // This code only updates the structural index in the parser, it does not update any json_iterator
   // instance.
   size_t remaining = len - _batch_start;
+  stage1_mode mode;
   if (remaining <= batch_size) {
-    return p.implementation->stage1(&buf[_batch_start], remaining, stage1_mode::streaming_final);
+    // Final batch
+    switch (format) {
+      case stream_format::json_sequence:
+        mode = stage1_mode::json_sequence_final;
+        break;
+      case stream_format::comma_delimited:
+        mode = stage1_mode::comma_delimited_final;
+        break;
+      default:
+        mode = stage1_mode::streaming_final;
+        break;
+    }
+    return p.implementation->stage1(&buf[_batch_start], remaining, mode);
   } else {
-    return p.implementation->stage1(&buf[_batch_start], batch_size, stage1_mode::streaming_partial);
+    // Partial batch
+    switch (format) {
+      case stream_format::json_sequence:
+        mode = stage1_mode::json_sequence_partial;
+        break;
+      case stream_format::comma_delimited:
+        mode = stage1_mode::comma_delimited_partial;
+        break;
+      default:
+        mode = stage1_mode::streaming_partial;
+        break;
+    }
+    return p.implementation->stage1(&buf[_batch_start], batch_size, mode);
   }
 }

@@ -181906,11 +234021,19 @@ simdjson_inline size_t document_stream::iterator::current_index() const noexcept
 }

 simdjson_inline std::string_view document_stream::iterator::source() const noexcept {
-  auto depth = stream->doc.iter.depth();
+  // On error (e.g., CAPACITY), there is no document to walk: return the rest of
+  // the input, as the DOM document_stream does.
+  if (stream->error) {
+    return std::string_view(reinterpret_cast<const char*>(stream->buf) + current_index(), stream->len - current_index());
+  }
+  // Always walk from the root of the document, whatever the current position
+  // of the document iterator: the user may have already consumed part of the
+  // document, so the iterator's current depth must not be used here.
+  depth_t depth = 1;
   auto cur_struct_index = stream->doc.iter._root - stream->parser->implementation->structural_indexes.get();

-  // If at root, process the first token to determine if scalar value
-  if (stream->doc.iter.at_root()) {
+  // Process the first token to determine if scalar value
+  {
     switch (stream->buf[stream->batch_start + stream->parser->implementation->structural_indexes[cur_struct_index]]) {
       case '{': case '[':   // Depth=1 already at start of document
         break;
@@ -181918,14 +234041,47 @@ simdjson_inline std::string_view document_stream::iterator::source() const noexc
         depth--;
         break;
       default:    // Scalar value document
-        // TODO: We could remove trailing whitespaces
         // This returns a string spanning from start of value to the beginning of the next document (excluded)
         {
           auto next_index = stream->batch_start + stream->parser->implementation->structural_indexes[++cur_struct_index];
           // normally the length would be next_index - current_index() - 1, except for the last document
           size_t svlen = next_index - current_index();
           const char *start = reinterpret_cast<const char*>(stream->buf) + current_index();
-          while(svlen > 1 && (std::isspace(start[svlen-1]) || start[svlen-1] == '\0')) {
+          // When the scalar is followed by a truncated document, the structural
+          // indexes of that document were dropped and next_index is the end of
+          // the input, so we bound the scalar by scanning the token itself.
+          size_t token_len = 0;
+          if (*start == '"') {
+            token_len = 1;
+            while (token_len < svlen) {
+              char c = start[token_len++];
+              if (c == '\\') {
+                token_len++;
+              } else if (c == '"') {
+                break;
+              }
+            }
+          } else {
+            while (token_len < svlen) {
+              char c = start[token_len];
+              if (std::isspace(static_cast<unsigned char>(c)) || c == ',' || c == '{' || c == '[' || c == '\0' || static_cast<uint8_t>(c) == 0x1E) {
+                break;
+              }
+              token_len++;
+            }
+          }
+          if (token_len > 0 && token_len < svlen) {
+            svlen = token_len;
+          }
+          // Trim trailing whitespace, NUL, and RS (0x1E). In RFC 7464
+          // json_sequence mode the scanner classifies RS as a scalar
+          // character, so an RS-prefixed scalar document (number / true /
+          // false / null / string) has no closing structural index and the
+          // slice runs all the way up to the next document's RS. RS cannot
+          // legally appear in a JSON value at the source level (control
+          // characters in strings must be escaped as \u001E), so stripping
+          // it is safe in every stream_format.
+          while(svlen > 1 && (std::isspace(static_cast<unsigned char>(start[svlen-1])) || start[svlen-1] == '\0' || static_cast<uint8_t>(start[svlen-1]) == 0x1E || (stream->format == stream_format::comma_delimited && start[svlen-1] == ','))) {
             svlen--;
           }
           return std::string_view(start, svlen);
@@ -182050,11 +234206,19 @@ simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> field::un
   return answer;
 }

+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_warn_unused simdjson_result<std::u8string_view> field::unescaped_u8key(bool allow_replacement) noexcept {
+  std::string_view key;
+  SIMDJSON_TRY( unescaped_key(allow_replacement).get(key) );
+  return internal::as_u8string_view(key);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 template <typename string_type>
 simdjson_inline simdjson_warn_unused error_code field::unescaped_key(string_type& receiver, bool allow_replacement) noexcept {
   std::string_view key;
   SIMDJSON_TRY( unescaped_key(allow_replacement).get(key) );
-  receiver = key;
+  internal::assign_utf8(receiver, key);
   return SUCCESS;
 }

@@ -182076,6 +234240,12 @@ simdjson_inline std::string_view field::escaped_key() const noexcept {
   return std::string_view(reinterpret_cast<const char*>(first.buf), end_quote - first.buf);
 }

+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline std::u8string_view field::escaped_u8key() const noexcept {
+  return internal::as_u8string_view(escaped_key());
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 simdjson_inline value &field::value() & noexcept {
   return second;
 }
@@ -182120,11 +234290,25 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<rvv_vls::ondem
   return first.escaped_key();
 }

+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<rvv_vls::ondemand::field>::escaped_u8key() noexcept {
+  if (error()) { return error(); }
+  return first.escaped_u8key();
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 simdjson_inline simdjson_result<std::string_view> simdjson_result<rvv_vls::ondemand::field>::unescaped_key(bool allow_replacement) noexcept {
   if (error()) { return error(); }
   return first.unescaped_key(allow_replacement);
 }

+#if SIMDJSON_SUPPORTS_CHAR8_T
+simdjson_inline simdjson_result<std::u8string_view> simdjson_result<rvv_vls::ondemand::field>::unescaped_u8key(bool allow_replacement) noexcept {
+  if (error()) { return error(); }
+  return first.unescaped_u8key(allow_replacement);
+}
+#endif // SIMDJSON_SUPPORTS_CHAR8_T
+
 template<typename string_type>
 simdjson_warn_unused simdjson_inline error_code simdjson_result<rvv_vls::ondemand::field>::unescaped_key(string_type &receiver, bool allow_replacement) noexcept {
   if (error()) { return error(); }
@@ -182168,6 +234352,9 @@ simdjson_inline json_iterator::json_iterator(json_iterator &&other) noexcept
     _depth{other._depth},
     _root{other._root},
     _streaming{other._streaming}
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+    , _allow_incomplete_json{other._allow_incomplete_json}
+#endif
 {
   other.parser = nullptr;
 }
@@ -182179,6 +234366,9 @@ simdjson_inline json_iterator &json_iterator::operator=(json_iterator &&other) n
   _depth = other._depth;
   _root = other._root;
   _streaming = other._streaming;
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  _allow_incomplete_json = other._allow_incomplete_json;
+#endif
   other.parser = nullptr;
   return *this;
 }
@@ -182205,7 +234395,8 @@ simdjson_inline json_iterator::json_iterator(const uint8_t *buf, ondemand::parse
       _string_buf_loc{parser->string_buf.get()},
       _depth{1},
       _root{parser->implementation->structural_indexes.get()},
-      _streaming{streaming}
+      _streaming{streaming},
+      _allow_incomplete_json{true}

 {
   logger::log_headers();
@@ -182277,7 +234468,8 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::skip_child(depth_
 #endif // SIMDJSON_CHECK_EOF
       break;
     case '"':
-      if(*peek() == ':') {
+      // At the end, peek() would read the sentinel, which points into the padding.
+      if(!at_end() && *peek() == ':') {
         // We are at a key!!!
         // This might happen if you just started an object and you skip it immediately.
         // Performance note: it would be nice to get rid of this check as it is somewhat
@@ -182320,7 +234512,7 @@ simdjson_warn_unused simdjson_inline error_code json_iterator::skip_child(depth_
     }
   }

-  return report_error(TAPE_ERROR, "not enough close braces");
+  return report_error(INCOMPLETE_ARRAY_OR_OBJECT, "not enough close braces");
 }

 SIMDJSON_POP_DISABLE_WARNINGS
@@ -182337,6 +234529,17 @@ simdjson_inline bool json_iterator::streaming() const noexcept {
   return _streaming;
 }

+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+simdjson_inline bool json_iterator::allow_incomplete_json() const noexcept {
+  return _allow_incomplete_json;
+}
+
+simdjson_inline size_t json_iterator::remaining_input_length(const uint8_t *json) const noexcept {
+  const uint8_t *end = token.buf + parser->_document_len;
+  return json < end ? size_t(end - json) : 0;
+}
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+
 simdjson_inline token_position json_iterator::root_position() const noexcept {
   return _root;
 }
@@ -182619,7 +234822,7 @@ inline std::ostream& operator<<(std::ostream& out, json_type type) noexcept {
         case json_type::string: out << "string"; break;
         case json_type::boolean: out << "boolean"; break;
         case json_type::null: out << "null"; break;
-        default: SIMDJSON_UNREACHABLE();
+        case json_type::unknown: out << "unknown"; break;
     }
     return out;
 }
@@ -182958,6 +235161,10 @@ inline void log_line(const json_iterator &iter, token_position index, depth_t de
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_iterator.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/value-inl.h" */
 /* amalgamation skipped (editor-only): #include "simdjson/jsonpathutil.h" */
+/* amalgamation skipped (editor-only): #include <utility> */
+/* amalgamation skipped (editor-only): #if SIMDJSON_SUPPORTS_CONCEPTS */
+/* amalgamation skipped (editor-only): #include <tuple> // std::forward_as_tuple/get for the variadic for_each adapter */
+/* amalgamation skipped (editor-only): #endif */
 /* amalgamation skipped (editor-only): #if SIMDJSON_STATIC_REFLECTION */
 /* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/json_string_builder.h"  // for constevalutil::fixed_string */
 /* amalgamation skipped (editor-only): #include <meta> */
@@ -182987,12 +235194,21 @@ simdjson_inline simdjson_result<value> object::find_field_unordered(const std::s
   return value(iter.child());
 }
 simdjson_inline simdjson_result<value> object::operator[](const std::string_view key) & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   return find_field_unordered(key);
 }
 simdjson_inline simdjson_result<value> object::operator[](const std::string_view key) && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   return std::forward<object>(*this).find_field_unordered(key);
 }
 simdjson_inline simdjson_result<value> object::find_field(const std::string_view key) & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   bool has_value;
   SIMDJSON_TRY( iter.find_field_raw(key).get(has_value) );
   if (!has_value) {
@@ -183002,6 +235218,9 @@ simdjson_inline simdjson_result<value> object::find_field(const std::string_view
   return value(iter.child());
 }
 simdjson_inline simdjson_result<value> object::find_field(const std::string_view key) && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   bool has_value;
   SIMDJSON_TRY( iter.find_field_raw(key).get(has_value) );
   if (!has_value) {
@@ -183011,6 +235230,150 @@ simdjson_inline simdjson_result<value> object::find_field(const std::string_view
   return value(iter.child());
 }

+#if SIMDJSON_SUPPORTS_CONCEPTS
+template <typename Selector, typename Func>
+  requires key_selector_type<Selector> &&
+           std::is_invocable_v<Func&, std::size_t, value>
+simdjson_flatten simdjson_inline for_each_result object::for_each(Func&& on_match)
+    noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, value>) {
+  // Single pass driven directly by the value_iterator, mirroring
+  // find_field_unordered_raw + value(iter.child()). Compared to walking via
+  // object_iterator/field, this avoids constructing a simdjson_result<field> and
+  // a field (key + value) for every field -- and the development-check bookkeeping
+  // in object_iterator -- building a value only for the fields that actually match.
+  // We operate on a copy of the iterator, as object::begin() would.
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  // Mirror object::begin(): for_each must start at the beginning of the object,
+  // not from some position left behind by a prior find_field on the same object.
+  if (!iter.is_at_iterator_start()) { return {OUT_OF_ORDER_ITERATION, 0}; }
+#endif
+  value_iterator it = iter;
+  std::size_t matched = 0;
+  // Track which selector indices have already matched, as a compile-time bitset
+  // (one 64-bit word per 64 keys): the callback fires once per key, on its first
+  // occurrence, and we stop as soon as every key has matched.
+  constexpr std::size_t seen_words = (Selector::size() + 63) / 64;
+  std::array<std::uint64_t, seen_words> seen{};
+  while (it.is_open()) {
+    raw_json_string key;
+    error_code error;
+    std::size_t idx;
+    if constexpr (Selector::window.ok) {
+      // A window selector confirms a key from its raw bytes alone (the closing
+      // quote bounds it), so we take the length-free path: field_key (no backward
+      // scan, mirroring the ordered find_field) feeding match_raw(raw_json_string).
+      if ((error = it.field_key().get(key))) { it.abandon(); return {error, matched}; }
+      if ((error = it.field_value())) { it.abandon(); return {error, matched}; }
+      idx = Selector::match_raw(key);
+    } else {
+      // Otherwise derive the key length from the structural index (the following
+      // ':' token) to feed the length-aware hash, avoiding a forward SIMD scan.
+      std::size_t key_len;
+      if ((error = it.field_key_with_length(key, key_len))) { it.abandon(); return {error, matched}; }
+      if ((error = it.field_value())) { it.abandon(); return {error, matched}; }
+      idx = Selector::match_raw(key.raw(), key_len);
+    }
+    if (idx < Selector::size()) {
+      const std::uint64_t seen_bit = std::uint64_t{1} << (idx & 63);
+      std::uint64_t &seen_word = seen[idx >> 6];
+      if (!(seen_word & seen_bit)) {
+        seen_word |= seen_bit;
+        value matched_value(it.child());
+        // The callback may return void or anything convertible to error_code
+        // (error_code itself, or a for_each_result from a nested for_each). When
+        // it yields an error_code, we stop at the first non-SUCCESS result and
+        // propagate it so the caller can surface value-parse errors (for example,
+        // a type mismatch on a matched field). A void-returning callback is
+        // responsible for handling its own errors.
+        if constexpr (std::is_convertible_v<decltype(on_match(idx, matched_value)), error_code>) {
+          // Unlike the internal-error paths above, a callback error does not
+          // abandon the iterator: we leave it recoverable so the caller can keep
+          // using the object (or its parent) after handling the error.
+          if ((error = on_match(idx, matched_value))) { return {error, matched}; }
+        } else {
+          on_match(idx, matched_value);
+        }
+        if (++matched >= Selector::size()) { break; }
+      }
+    }
+    // Mirror object_iterator::operator++'s safety rail: if the callback consumed
+    // the value and left the iterator closed or in error (e.g. a void callback
+    // that swallowed a fatal sub-iteration error), stop here rather than calling
+    // skip_child on a closed iterator.
+    if (!it.is_open()) { break; }
+    // Skip the value (a no-op if the callback consumed it) and step to the next
+    // field; has_next_field() ends the container on '}', which closes the loop.
+    if ((error = it.skip_child())) { it.abandon(); return {error, matched}; }
+    if ((error = it.has_next_field().error())) { return {error, matched}; }
+  }
+  return {SUCCESS, matched};
+}
+
+namespace key_selector_for_each_detail {
+
+// Dispatch a matched value to the I-th handler in the tuple (0-based). A handler
+// is either an invocable taking the value (void- or error_code-returning) or a
+// deserialization target, in which case we assign via value::get. We always
+// return error_code so the core (index, value) for_each can treat the adapter
+// uniformly.
+template <typename Tuple, std::size_t... Is>
+simdjson_really_inline error_code dispatch_value(
+    std::size_t idx, Tuple& handlers, value v, std::index_sequence<Is...>) {
+  error_code err = SUCCESS;
+  auto try_one = [&](auto Ic) {
+    constexpr std::size_t I = decltype(Ic)::value;
+    if (idx == I) {
+      auto&& h = std::get<I>(handlers);
+      using H = std::remove_reference_t<decltype(h)>;
+      if constexpr (std::is_invocable_v<H&, value>) {
+        // A handler returning void runs for its side effects; one returning
+        // anything convertible to error_code (error_code, or a for_each_result
+        // from a nested for_each) has its error captured and propagated.
+        if constexpr (std::is_convertible_v<decltype(h(v)), error_code>) {
+          err = h(v);
+        } else {
+          h(v);
+        }
+      } else {
+        // Direct deserialization target: assign the matched value into it.
+        err = v.get(h);
+      }
+    }
+  };
+  (try_one(std::integral_constant<std::size_t, Is>{}), ...);
+  return err;
+}
+
+} // namespace key_selector_for_each_detail
+
+template <typename Selector, typename... Handlers>
+  requires key_selector_type<Selector> &&
+           (sizeof...(Handlers) == Selector::size()) &&
+           (key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_flatten simdjson_inline for_each_result object::for_each(Handlers&&... on_match)
+    noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+  auto handlers = std::forward_as_tuple(std::forward<Handlers>(on_match)...);
+  // Reuse the single (index, value) implementation via a tiny adapter.
+  // The adapter is called once per *matched* key (very few); the hot path
+  // (iteration + match_raw + seen bitset) stays exactly the same.
+  return this->template for_each<Selector>(
+      [&](std::size_t i, value v) -> error_code {
+        return key_selector_for_each_detail::dispatch_value(
+            i, handlers, v, std::make_index_sequence<Selector::size()>{});
+      });
+}
+
+template <constevalutil::fixed_string... Keys, typename... Handlers>
+  requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+           (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+           (key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline for_each_result object::for_each(Handlers&&... on_match)
+    noexcept(key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+  using Selector = key_selector<Keys...>;
+  return this->template for_each<Selector>(std::forward<Handlers>(on_match)...);
+}
+#endif
+
 simdjson_inline simdjson_result<object> object::start(value_iterator &iter) noexcept {
   SIMDJSON_TRY( iter.start_object().error() );
   return object(iter);
@@ -183046,6 +235409,9 @@ simdjson_warn_unused simdjson_inline error_code object::consume() noexcept {
 }

 simdjson_inline simdjson_result<std::string_view> object::raw_json() noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   const uint8_t * starting_point{iter.peek_start()};
   auto error = consume();
   if(error) { return error; }
@@ -183067,9 +235433,17 @@ simdjson_inline object::object(const value_iterator &_iter) noexcept
 {
 }

-simdjson_inline simdjson_result<object_iterator> object::begin() noexcept {
+simdjson_inline simdjson_result<object_iterator> object::begin() & noexcept {
 #if SIMDJSON_DEVELOPMENT_CHECKS
-  if (!iter.is_at_iterator_start()) { return OUT_OF_ORDER_ITERATION; }
+  if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
+  return object_iterator(iter, this);
+#endif
+  return object_iterator(iter);
+}
+simdjson_inline simdjson_result<object_iterator> object::begin() && noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  // The object is a temporary that the iterator may outlive: do not lock it.
+  if (!iter.is_at_iterator_start() || locked) { return OUT_OF_ORDER_ITERATION; }
 #endif
   return object_iterator(iter);
 }
@@ -183078,7 +235452,9 @@ simdjson_inline simdjson_result<object_iterator> object::end() noexcept {
 }

 inline simdjson_result<value> object::at_pointer(std::string_view json_pointer) noexcept {
-  if (json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
+  // An empty pointer has no json_pointer[0]: with a default-constructed
+  // std::string_view, reading it dereferences a null pointer.
+  if (json_pointer.empty() || json_pointer[0] != '/') { return INVALID_JSON_POINTER; }
   json_pointer = json_pointer.substr(1);
   size_t slash = json_pointer.find('/');
   std::string_view key = json_pointer.substr(0, slash);
@@ -183180,6 +235556,9 @@ simdjson_inline simdjson_result<size_t> object::count_fields() & noexcept {
 }

 simdjson_inline simdjson_result<bool> object::is_empty() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   bool is_not_empty;
   auto error = iter.reset_object().get(is_not_empty);
   if(error) { return error; }
@@ -183187,9 +235566,41 @@ simdjson_inline simdjson_result<bool> object::is_empty() & noexcept {
 }

 simdjson_inline simdjson_result<bool> object::reset() & noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
   return iter.reset_object();
 }

+simdjson_inline object_position object::get_current_position() const noexcept {
+  return object_position{iter.position(), iter.json_iter().depth()};
+}
+
+simdjson_inline error_code object::revert_position(object_position position) noexcept {
+#if SIMDJSON_DEVELOPMENT_CHECKS
+  if (locked) return OUT_OF_ORDER_ITERATION;
+#endif
+  // json_iterator::reenter_child() requires the live depth to be exactly
+  // one level shallower than the target (matching how every other depth
+  // transition in this iterator works), and under SIMDJSON_DEVELOPMENT_CHECKS
+  // additionally validates against the parser's per-depth container-start
+  // bookkeeping. Neither applies here: depending on what was captured and
+  // what has happened since (a scalar field fully consumed, a compound
+  // value left open, a find_field() miss that scanned past everything),
+  // the live depth when reverting can be any number of levels away from
+  // the captured one, and the captured depth is not necessarily a
+  // container's own start. reenter_at() moves directly, matching how
+  // reset_object() itself repositions without going through reenter_child().
+  iter.reenter_at(position.position, position.depth);
+  return SUCCESS;
+}
+
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline void object::set_locked(bool _locked) noexcept {
+  locked = _locked;
+}
+#endif
+
 #if SIMDJSON_SUPPORTS_CONCEPTS && SIMDJSON_STATIC_REFLECTION

 template<constevalutil::fixed_string... FieldNames, typename T>
@@ -183247,10 +235658,14 @@ simdjson_inline simdjson_result<rvv_vls::ondemand::object>::simdjson_result(rvv_
 simdjson_inline simdjson_result<rvv_vls::ondemand::object>::simdjson_result(error_code error) noexcept
     : implementation_simdjson_result_base<rvv_vls::ondemand::object>(error) {}

-simdjson_inline simdjson_result<rvv_vls::ondemand::object_iterator> simdjson_result<rvv_vls::ondemand::object>::begin() noexcept {
+simdjson_inline simdjson_result<rvv_vls::ondemand::object_iterator> simdjson_result<rvv_vls::ondemand::object>::begin() & noexcept {
   if (error()) { return error(); }
   return first.begin();
 }
+simdjson_inline simdjson_result<rvv_vls::ondemand::object_iterator> simdjson_result<rvv_vls::ondemand::object>::begin() && noexcept {
+  if (error()) { return error(); }
+  return std::move(first).begin();
+}
 simdjson_inline simdjson_result<rvv_vls::ondemand::object_iterator> simdjson_result<rvv_vls::ondemand::object>::end() noexcept {
   if (error()) { return error(); }
   return first.end();
@@ -183304,11 +235719,55 @@ simdjson_inline error_code simdjson_result<rvv_vls::ondemand::object>::for_each_
   return first.for_each_at_path_with_wildcard(json_path, std::forward<Func>(callback));
 }

+#if SIMDJSON_SUPPORTS_CONCEPTS
+template <typename Selector, typename Func>
+  requires rvv_vls::ondemand::key_selector_type<Selector> &&
+           std::is_invocable_v<Func&, std::size_t, rvv_vls::ondemand::value>
+simdjson_inline rvv_vls::ondemand::for_each_result
+simdjson_result<rvv_vls::ondemand::object>::for_each(Func&& on_match)
+    noexcept(std::is_nothrow_invocable_v<Func&, std::size_t, rvv_vls::ondemand::value>) {
+  if (error()) { return {error(), 0}; }
+  return first.template for_each<Selector>(std::forward<Func>(on_match));
+}
+
+template <typename Selector, typename... Handlers>
+  requires rvv_vls::ondemand::key_selector_type<Selector> &&
+           (sizeof...(Handlers) == Selector::size()) &&
+           (rvv_vls::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline rvv_vls::ondemand::for_each_result
+simdjson_result<rvv_vls::ondemand::object>::for_each(Handlers&&... on_match)
+    noexcept(rvv_vls::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+  if (error()) { return {error(), 0}; }
+  return first.template for_each<Selector>(std::forward<Handlers>(on_match)...);
+}
+
+template <constevalutil::fixed_string... Keys, typename... Handlers>
+  requires (sizeof...(Handlers) == sizeof...(Keys)) &&
+           (sizeof...(Keys) >= 1) && (sizeof...(Keys) <= 255) &&
+           (rvv_vls::ondemand::key_selector_for_each_detail::field_handler<Handlers> && ...)
+simdjson_inline rvv_vls::ondemand::for_each_result
+simdjson_result<rvv_vls::ondemand::object>::for_each(Handlers&&... on_match)
+    noexcept(rvv_vls::ondemand::key_selector_for_each_detail::nothrow_field_handlers_v<Handlers...>) {
+  if (error()) { return {error(), 0}; }
+  return first.template for_each<Keys...>(std::forward<Handlers>(on_match)...);
+}
+#endif
+
 inline simdjson_result<bool> simdjson_result<rvv_vls::ondemand::object>::reset() noexcept {
   if (error()) { return error(); }
   return first.reset();
 }

+inline simdjson_result<rvv_vls::ondemand::object_position> simdjson_result<rvv_vls::ondemand::object>::get_current_position() noexcept {
+  if (error()) { return error(); }
+  return first.get_current_position();
+}
+
+inline error_code simdjson_result<rvv_vls::ondemand::object>::revert_position(rvv_vls::ondemand::object_position position) noexcept {
+  if (error()) { return error(); }
+  return first.revert_position(position);
+}
+
 inline simdjson_result<bool> simdjson_result<rvv_vls::ondemand::object>::is_empty() noexcept {
   if (error()) { return error(); }
   return first.is_empty();
@@ -183352,6 +235811,61 @@ simdjson_inline object_iterator::object_iterator(const value_iterator &_iter) no
   : iter{_iter}
 {}

+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline object_iterator::object_iterator(const value_iterator &_iter, object* _parent) noexcept
+  : parent{_parent}, iter{_iter}
+{
+  if (parent) parent->set_locked(true);
+}
+#endif
+
+#if SIMDJSON_DEVELOPMENT_CHECKS
+simdjson_inline object_iterator::~object_iterator() noexcept
+{
+  if (parent) parent->set_locked(false);
+}
+
+simdjson_inline object_iterator::object_iterator(object_iterator&& other) noexcept
+  : has_been_referenced{other.has_been_referenced},
+    parent{other.parent},
+    iter{std::move(other.iter)}
+{
+  other.parent = nullptr;
+}
+
+simdjson_inline object_iterator& object_iterator::operator=(object_iterator&& other) noexcept {
+  if (this != &other)
+  {
+    if (parent)
+      parent->set_locked(false);
+    has_been_referenced = other.has_been_referenced;
+    parent = other.parent;
+    iter = std::move(other.iter);
+
+    other.parent = nullptr;
+  }
+  return *this;
+}
+
+simdjson_inline object_iterator::object_iterator(const object_iterator& other) noexcept
+  : has_been_referenced{other.has_been_referenced},
+    parent{nullptr},
+    iter{other.iter}
+{}
+
+simdjson_inline object_iterator& object_iterator::operator=(const object_iterator& other) noexcept {
+  if (this != &other)
+  {
+    if (parent)
+      parent->set_locked(false);
+    has_been_referenced = other.has_been_referenced;
+    parent = nullptr;
+    iter = other.iter;
+  }
+  return *this;
+}
+#endif
+
 simdjson_inline simdjson_result<field> object_iterator::operator*() noexcept {
 #if SIMDJSON_DEVELOPMENT_CHECKS
   // We must call * once per iteration.
@@ -183479,6 +235993,147 @@ simdjson_inline simdjson_result<rvv_vls::ondemand::object_iterator> &simdjson_re

 #endif // SIMDJSON_GENERIC_ONDEMAND_OBJECT_ITERATOR_INL_H
 /* end file simdjson/generic/ondemand/object_iterator-inl.h for rvv_vls */
+/* including simdjson/generic/ondemand/ranges-inl.h for rvv_vls: #include "simdjson/generic/ondemand/ranges-inl.h" */
+/* begin file simdjson/generic/ondemand/ranges-inl.h for rvv_vls */
+#ifndef SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H
+
+/* amalgamation skipped (editor-only): #ifndef SIMDJSON_CONDITIONAL_INCLUDE */
+/* amalgamation skipped (editor-only): #define SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/base.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/ranges.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/array_iterator-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object-inl.h" */
+/* amalgamation skipped (editor-only): #include "simdjson/generic/ondemand/object_iterator-inl.h" */
+/* amalgamation skipped (editor-only): #endif // SIMDJSON_CONDITIONAL_INCLUDE */
+
+#if SIMDJSON_SUPPORTS_RANGES
+
+namespace simdjson {
+namespace rvv_vls {
+namespace ondemand {
+
+//
+// array_range_iterator
+//
+
+simdjson_inline array_range_iterator::array_range_iterator(array_iterator iter) noexcept
+  : iter_{iter} {}
+
+simdjson_inline simdjson_result<value> array_range_iterator::operator*() const noexcept {
+  return *iter_;
+}
+
+simdjson_inline array_range_iterator& array_range_iterator::operator++() noexcept {
+  ++iter_;
+  return *this;
+}
+
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+simdjson_inline void array_range_iterator::operator++(int) noexcept {
+  ++*this;
+}
+SIMDJSON_POP_DISABLE_WARNINGS
+
+//
+// array_range
+//
+
+simdjson_inline array_range::array_range(array& arr) noexcept {
+  auto b = arr.begin();
+  if (b.error()) { error_ = b.error(); return; }
+  begin_ = b.value_unsafe();
+  end_ = arr.end().value_unsafe();
+}
+
+simdjson_inline array_range_iterator array_range::begin() noexcept {
+  return array_range_iterator(begin_);
+}
+
+simdjson_inline array_range_iterator array_range::end() noexcept {
+  return array_range_iterator(end_);
+}
+
+//
+// object_range_iterator
+//
+
+simdjson_inline object_range_iterator::object_range_iterator(object_iterator iter) noexcept
+  : iter_{iter} {}
+
+simdjson_inline simdjson_result<field> object_range_iterator::operator*() const noexcept {
+  return *iter_;
+}
+
+simdjson_inline object_range_iterator& object_range_iterator::operator++() noexcept {
+  ++iter_;
+  return *this;
+}
+
+SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
+simdjson_inline void object_range_iterator::operator++(int) noexcept {
+  ++*this;
+}
+SIMDJSON_POP_DISABLE_WARNINGS
+
+
+//
+// object_range
+//
+
+simdjson_inline object_range::object_range(object& obj) noexcept {
+  auto b = obj.begin();
+  if (b.error()) { error_ = b.error(); return; }
+  begin_ = b.value_unsafe();
+  end_ = obj.end().value_unsafe();
+}
+
+simdjson_inline object_range_iterator object_range::begin() noexcept {
+  return object_range_iterator(begin_);
+}
+
+simdjson_inline object_range_iterator object_range::end() noexcept {
+  return object_range_iterator(end_);
+}
+
+//
+// Free functions
+//
+
+simdjson_inline array_range get_range(array& arr) noexcept {
+  return array_range(arr);
+}
+
+simdjson_inline object_range get_key_value_range(object& obj) noexcept {
+  return object_range(obj);
+}
+
+#if SIMDJSON_EXCEPTIONS
+simdjson_inline array_range get_range(simdjson_result<array> result) {
+  return array_range(result.value());
+}
+
+simdjson_inline object_range get_key_value_range(simdjson_result<object> result) {
+  return object_range(result.value());
+}
+#endif // SIMDJSON_EXCEPTIONS
+
+} // namespace ondemand
+} // namespace rvv_vls
+} // namespace simdjson
+
+// Verify the range wrapper types satisfy the expected C++20 concepts.
+static_assert(std::input_iterator<simdjson::rvv_vls::ondemand::array_range_iterator>);
+static_assert(std::input_iterator<simdjson::rvv_vls::ondemand::object_range_iterator>);
+static_assert(std::ranges::input_range<simdjson::rvv_vls::ondemand::array_range>);
+static_assert(std::ranges::input_range<simdjson::rvv_vls::ondemand::object_range>);
+static_assert(std::ranges::view<simdjson::rvv_vls::ondemand::array_range>);
+static_assert(std::ranges::view<simdjson::rvv_vls::ondemand::object_range>);
+
+#endif // SIMDJSON_SUPPORTS_RANGES
+
+#endif // SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H
+/* end file simdjson/generic/ondemand/ranges-inl.h for rvv_vls */
 /* including simdjson/generic/ondemand/parser-inl.h for rvv_vls: #include "simdjson/generic/ondemand/parser-inl.h" */
 /* begin file simdjson/generic/ondemand/parser-inl.h for rvv_vls */
 #ifndef SIMDJSON_GENERIC_ONDEMAND_PARSER_INL_H
@@ -183510,7 +236165,10 @@ simdjson_warn_unused simdjson_inline error_code parser::allocate(size_t new_capa

   // string_capacity copied from document::allocate
   _capacity = 0;
-  size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * new_capacity / 3 + SIMDJSON_PADDING, 64);
+  if(5 * (new_capacity / 3) + SIMDJSON_PADDING < SIMDJSON_PADDING) {
+    return CAPACITY; // overflow, only happen on legacy 32-bit systems with very large capacity
+  }
+  size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * (new_capacity / 3) + SIMDJSON_PADDING, 64);
   string_buf.reset(new (std::nothrow) uint8_t[string_capacity]);
 #if SIMDJSON_DEVELOPMENT_CHECKS
   start_positions.reset(new (std::nothrow) token_position[new_max_depth]);
@@ -183535,6 +236193,7 @@ simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate(p
   if (!json.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }

   json.remove_utf8_bom();
+  _document_len = json.length();

   // Allocate if needed
   if (capacity() < json.length() || !string_buf) {
@@ -183551,6 +236210,7 @@ simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate_a
   if (!json.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }

   json.remove_utf8_bom();
+  _document_len = json.length();

   // Allocate if needed
   if (capacity() < json.length() || !string_buf) {
@@ -183616,6 +236276,34 @@ simdjson_warn_unused simdjson_inline simdjson_result<json_iterator> parser::iter
   return json_iterator(reinterpret_cast<const uint8_t *>(json.data()), this);
 }

+inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size) noexcept {
+  if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
+  if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+    buf += 3;
+    len -= 3;
+  }
+  return document_stream(*this, buf, len, batch_size, false, stream_format::whitespace_delimited);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size) noexcept {
+  return iterate_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(padded_string_view s, size_t batch_size) noexcept {
+  if (!s.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
+  return iterate_many(s.data(), s.length(), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const padded_string &s, size_t batch_size) noexcept {
+  return iterate_many(padded_string_view(s), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const std::string &s, size_t batch_size) noexcept {
+  return iterate_many(padded_string_view(s), batch_size);
+}
+inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size) noexcept {
+  return iterate_many(pad(s), batch_size);
+}
+
+#ifndef SIMDJSON_DISABLE_DEPRECATED_API
+SIMDJSON_PUSH_DISABLE_WARNINGS
+SIMDJSON_DISABLE_DEPRECATED_WARNING
 inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept {
   // Warning: no check is done on the buffer padding. We trust the user.
   if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
@@ -183623,8 +236311,11 @@ inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf,
     buf += 3;
     len -= 3;
   }
-  if(allow_comma_separated && batch_size < len) { batch_size = len; }
-  return document_stream(*this, buf, len, batch_size, allow_comma_separated);
+  // Map allow_comma_separated to stream_format::comma_delimited
+  if (allow_comma_separated) {
+    return document_stream(*this, buf, len, batch_size, false, stream_format::comma_delimited);
+  }
+  return document_stream(*this, buf, len, batch_size, false, stream_format::whitespace_delimited);
 }

 inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size, bool allow_comma_separated) noexcept {
@@ -183644,6 +236335,51 @@ inline simdjson_result<document_stream> parser::iterate_many(const std::string &
 inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size, bool allow_comma_separated) noexcept {
   return iterate_many(pad(s), batch_size, allow_comma_separated);
 }
+SIMDJSON_POP_DISABLE_WARNINGS
+#endif // SIMDJSON_DISABLE_DEPRECATED_API
+
+inline simdjson_result<document_stream> parser::iterate_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+  if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
+  if((len >= 3) && (std::memcmp(buf, "\xEF\xBB\xBF", 3) == 0)) {
+    buf += 3;
+    len -= 3;
+  }
+  if (format == stream_format::comma_delimited_array) {
+    // Strip leading JSON whitespace.
+    while (len > 0 && (buf[0] == ' ' || buf[0] == '\t' || buf[0] == '\n' || buf[0] == '\r')) {
+      buf++; len--;
+    }
+    // Expect the opening '['.
+    if (len == 0 || buf[0] != '[') { return TAPE_ERROR; }
+    buf++; len--;
+    // Strip trailing JSON whitespace.
+    while (len > 0 && (buf[len-1] == ' ' || buf[len-1] == '\t' || buf[len-1] == '\n' || buf[len-1] == '\r')) {
+      len--;
+    }
+    // Expect the closing ']'.
+    if (len == 0 || buf[len-1] != ']') { return TAPE_ERROR; }
+    len--;
+    // Fall through to comma_delimited over the array contents.
+    format = stream_format::comma_delimited;
+  }
+  return document_stream(*this, buf, len, batch_size, false, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const char *buf, size_t len, size_t batch_size, stream_format format) noexcept {
+  return iterate_many(reinterpret_cast<const uint8_t *>(buf), len, batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(padded_string_view s, size_t batch_size, stream_format format) noexcept {
+  if (!s.has_sufficient_padding()) { return INSUFFICIENT_PADDING; }
+  return iterate_many(s.data(), s.length(), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const std::string &s, size_t batch_size, stream_format format) noexcept {
+  return iterate_many(padded_string_view(s), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(std::string &s, size_t batch_size, stream_format format) noexcept {
+  return iterate_many(pad(s), batch_size, format);
+}
+inline simdjson_result<document_stream> parser::iterate_many(const padded_string &s, size_t batch_size, stream_format format) noexcept {
+  return iterate_many(padded_string_view(s), batch_size, format);
+}
 simdjson_pure simdjson_inline size_t parser::capacity() const noexcept {
   return _capacity;
 }
@@ -184051,6 +236787,27 @@ namespace simdjson {
 namespace rvv_vls {
 namespace ondemand {

+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+simdjson_inline bool raw_json_string_is_quote_terminated(const uint8_t *json, uint32_t max_len) noexcept {
+  bool escaping{false};
+  for (uint32_t i = 1; i < max_len; i++) {
+    switch (json[i]) {
+      case '"':
+        if (!escaping) { return true; }
+        escaping = false;
+        break;
+      case '\\':
+        escaping = !escaping;
+        break;
+      default:
+        escaping = false;
+        break;
+    }
+  }
+  return false;
+}
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+
 simdjson_inline value_iterator::value_iterator(
   json_iterator *json_iter,
   depth_t depth,
@@ -184438,6 +237195,22 @@ simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iter
   return raw_json_string(key);
 }

+simdjson_warn_unused simdjson_inline error_code value_iterator::field_key_with_length(raw_json_string &key, std::size_t &len) noexcept {
+  assert_at_next();
+
+  const uint8_t *k = _json_iter->return_current_and_advance();
+  if (*(k++) != '"') { return report_error(TAPE_ERROR, "Object key is not a string"); }
+  // After return_current_and_advance(), the current token is the ':' that follows
+  // the key. The closing quote sits just before it (only JSON whitespace may
+  // intervene), so step back from the ':' to the closing quote to get the length.
+  // In minified JSON this is a single back-step.
+  const char *q = reinterpret_cast<const char *>(_json_iter->peek());
+  do { --q; } while (*q != '"');
+  key = raw_json_string(k);
+  len = static_cast<std::size_t>(q - reinterpret_cast<const char *>(k));
+  return SUCCESS;
+}
+
 simdjson_warn_unused simdjson_inline error_code value_iterator::field_value() noexcept {
   assert_at_next();

@@ -184555,7 +237328,7 @@ simdjson_warn_unused simdjson_inline error_code value_iterator::get_string(strin
   auto saved_string_buf_loc = _json_iter->string_buf_loc();
   auto err = get_string(allow_replacement).get(content);
   if (err) { return err; }
-  receiver = content;
+  internal::assign_utf8(receiver, content);
   // Restore the string buffer location, effectively discarding any temporary string storage
   _json_iter->string_buf_loc() = saved_string_buf_loc;
   return SUCCESS;
@@ -184566,6 +237339,15 @@ simdjson_warn_unused simdjson_inline simdjson_result<std::string_view> value_ite
 simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iterator::get_raw_json_string() noexcept {
   auto json = peek_scalar("string");
   if (*json != '"') { return incorrect_type_error("Not a string"); }
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  if (_json_iter->allow_incomplete_json()) {
+    const size_t remaining_input_length = _json_iter->remaining_input_length(json);
+    const uint32_t max_len = remaining_input_length < peek_start_length() ? uint32_t(remaining_input_length) : peek_start_length();
+    if (!raw_json_string_is_quote_terminated(json, max_len)) {
+      return STRING_ERROR;
+    }
+  }
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
   advance_scalar("string");
   return raw_json_string(json+1);
 }
@@ -184599,6 +237381,16 @@ simdjson_warn_unused simdjson_inline simdjson_result<double> value_iterator::get
   if(result.error() == SUCCESS) { advance_non_root_scalar("double"); }
   return result;
 }
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_float() noexcept {
+  auto result = numberparsing::parse_float(peek_non_root_scalar("float"));
+  if(result.error() == SUCCESS) { advance_non_root_scalar("float"); }
+  return result;
+}
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_float_in_string() noexcept {
+  auto result = numberparsing::parse_float_in_string(peek_non_root_scalar("float"));
+  if(result.error() == SUCCESS) { advance_non_root_scalar("float"); }
+  return result;
+}
 simdjson_warn_unused simdjson_inline simdjson_result<bool> value_iterator::get_bool() noexcept {
   auto result = parse_bool(peek_non_root_scalar("bool"));
   if(result.error() == SUCCESS) { advance_non_root_scalar("bool"); }
@@ -184701,7 +237493,7 @@ simdjson_warn_unused simdjson_inline error_code value_iterator::get_root_string(
   auto saved_string_buf_loc = _json_iter->string_buf_loc();
   auto err = get_root_string(check_trailing, allow_replacement).get(content);
   if (err) { return err; }
-  receiver = content;
+  internal::assign_utf8(receiver, content);
   // Restore the string buffer location, effectively discarding any temporary string storage
   _json_iter->string_buf_loc() = saved_string_buf_loc;
   return SUCCESS;
@@ -184713,6 +237505,15 @@ simdjson_warn_unused simdjson_inline simdjson_result<raw_json_string> value_iter
   auto json = peek_scalar("string");
   if (*json != '"') { return incorrect_type_error("Not a string"); }
   if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
+  if (_json_iter->allow_incomplete_json()) {
+    const size_t remaining_input_length = _json_iter->remaining_input_length(json);
+    const uint32_t max_len = remaining_input_length < peek_root_length() ? uint32_t(remaining_input_length) : peek_root_length();
+    if (!raw_json_string_is_quote_terminated(json, max_len)) {
+      return STRING_ERROR;
+    }
+  }
+#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
   advance_scalar("string");
   return raw_json_string(json+1);
 }
@@ -184822,6 +237623,43 @@ simdjson_warn_unused simdjson_inline simdjson_result<double> value_iterator::get
   return result;
 }

+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_root_float(bool check_trailing) noexcept {
+  auto max_len = peek_root_length();
+  auto json = peek_root_scalar("float");
+  // We use the same buffer size as get_root_double: the number of significant
+  // digits that matter is smaller for binary32, but the JSON document may still
+  // spell out a long number that we must parse (and round) faithfully.
+  uint8_t tmpbuf[1074+8+1+1]; // +1 for null termination.
+  tmpbuf[1074+8+1] = '\0'; // make sure that buffer is always null terminated.
+  if (!_json_iter->copy_to_buffer(json, max_len, tmpbuf, 1074+8+1)) {
+    logger::log_error(*_json_iter, start_position(), depth(), "Root number more than 1082 characters");
+    return NUMBER_ERROR;
+  }
+  auto result = numberparsing::parse_float(tmpbuf);
+  if(result.error() == SUCCESS) {
+    if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+    advance_root_scalar("float");
+  }
+  return result;
+}
+
+simdjson_warn_unused simdjson_inline simdjson_result<float> value_iterator::get_root_float_in_string(bool check_trailing) noexcept {
+  auto max_len = peek_root_length();
+  auto json = peek_root_scalar("float");
+  uint8_t tmpbuf[1074+8+1+1]; // +1 for null termination.
+  tmpbuf[1074+8+1] = '\0'; // make sure that buffer is always null terminated.
+  if (!_json_iter->copy_to_buffer(json, max_len, tmpbuf, 1074+8+1)) {
+    logger::log_error(*_json_iter, start_position(), depth(), "Root number more than 1082 characters");
+    return NUMBER_ERROR;
+  }
+  auto result = numberparsing::parse_float_in_string(tmpbuf);
+  if(result.error() == SUCCESS) {
+    if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
+    advance_root_scalar("float");
+  }
+  return result;
+}
+
 simdjson_warn_unused simdjson_inline simdjson_result<bool> value_iterator::get_root_bool(bool check_trailing) noexcept {
   auto max_len = peek_root_length();
   auto json = peek_root_scalar("bool");
@@ -185050,6 +237888,21 @@ simdjson_inline void value_iterator::move_at_container_start() noexcept {
   _json_iter->token.set_position(_start_position + 1);
 }

+simdjson_inline void value_iterator::reenter_at(token_position position, depth_t depth) noexcept {
+  // Unlike reenter_child(), this does not require the live depth to be
+  // exactly one level shallower than depth, nor does it validate against
+  // the parser's per-depth container-start bookkeeping: neither holds in
+  // general for a caller-supplied snapshot (see object_position). What
+  // must still always hold, regardless of what was captured or how far
+  // the live iterator has since moved, is that position and depth are
+  // themselves sane values -- this is the same bound reenter_child()
+  // itself applies unconditionally.
+  SIMDJSON_ASSUME(position != nullptr);
+  SIMDJSON_ASSUME(depth >= 1 && depth < INT32_MAX);
+  _json_iter->_depth = depth;
+  _json_iter->token.set_position(position);
+}
+
 simdjson_inline simdjson_result<bool> value_iterator::reset_array() noexcept {
   if(error()) { return error(); }
   move_at_container_start();
@@ -186430,12 +239283,12 @@ public:
 	explicit auto_parser(std::remove_pointer_t<parser_type> &parser, padded_string_view const str) noexcept requires(std::is_pointer_v<parser_type>);
 	explicit auto_parser(padded_string_view const str) noexcept requires(std::is_pointer_v<parser_type>);
 	explicit auto_parser(parser_type parser, ondemand::document &&doc) noexcept requires(std::is_pointer_v<parser_type>);
-		auto_parser(auto_parser const &) = delete;
-		auto_parser &operator=(auto_parser const &) = delete;
-		auto_parser(auto_parser &&) noexcept = default;
-		auto_parser &operator=(auto_parser &&) noexcept = default;
-		~auto_parser() = default;
-
+	auto_parser(auto_parser const &) = delete;
+	auto_parser &operator=(auto_parser const &) = delete;
+	~auto_parser() = default;
+    // Prevent moving
+    auto_parser(auto_parser&&) = delete;
+	auto_parser &operator=(auto_parser &&) noexcept = delete;
 	simdjson_warn_unused std::remove_pointer_t<parser_type> &parser() noexcept;

 	template <typename T>
@@ -186469,11 +239322,6 @@ struct to_adaptor {
 	T operator()(simdjson_result<ondemand::value> &val) const noexcept;
 	auto operator()(padded_string_view const str) const noexcept;
 	auto operator()(ondemand::parser &parser, padded_string_view const str) const noexcept;
-	// The std::string is padded with reserve to ensure there is enough space for padding.
-	// Some sanitizers may not like this, so you can use simdjson::pad instead.
-	// simdjson::from(simdjson::pad(str))
-	auto operator()(std::string str) const noexcept;
-	auto operator()(ondemand::parser &parser, std::string str) const noexcept;
 };
 // deduction guide
 auto_parser(padded_string_view const str) -> auto_parser<ondemand::parser*>;
@@ -186484,7 +239332,12 @@ auto_parser(padded_string_view const str) -> auto_parser<ondemand::parser*>;
  * The simdjson::from instance is EXPERIMENTAL AND SUBJECT TO CHANGES.
  *
  * The `from` instance is a utility adaptor for parsing JSON strings into objects.
- * It provides a convenient way to convert JSON data into C++ objects using the `auto_parser`.
+ *
+ * The string must be a simdjson::padded_string_view, which can be created from a std::string
+ * with simdjson::pad(), from a simdjson::padded_string, or string literal using the `_padded`
+ * user-defined literal.
+ *
+ * The `from` instance provides a convenient way to convert JSON data into C++ objects using the `auto_parser`.
  *
  * Example usage:
  *
@@ -186555,9 +239408,6 @@ inline auto_parser<parser_type>::auto_parser(parser_type parser, ondemand::docum
   : auto_parser{*parser, std::move(doc)} {}


-
-
-
 template <typename parser_type>
 inline std::remove_pointer_t<parser_type> &auto_parser<parser_type>::parser() noexcept {
   if constexpr (std::is_pointer_v<parser_type>) {
@@ -186636,16 +239486,6 @@ template <typename T>
 inline auto to_adaptor<T>::operator()(ondemand::parser &parser, padded_string_view const str) const noexcept {
   return auto_parser<ondemand::parser *>{parser, str};
 }
-
-template <typename T>
-inline auto to_adaptor<T>::operator()(std::string str) const noexcept {
-  return auto_parser<ondemand::parser *>{pad_with_reserve(str)};
-}
-
-template <typename T>
-inline auto to_adaptor<T>::operator()(ondemand::parser &parser, std::string str) const noexcept {
-  return auto_parser<ondemand::parser *>{parser, pad_with_reserve(str)};
-}
 } // namespace internal
 } // namespace convert
 } // namespace simdjson
@@ -186721,14 +239561,17 @@ namespace compile_time {
 template <constevalutil::fixed_string json_str> consteval auto parse_json();

 } // namespace compile_time
-} // namespace simdjson

+inline namespace literals {

 template <simdjson::constevalutil::fixed_string str>
 consteval auto operator ""_json() {
   return simdjson::compile_time::parse_json<str>();
 }

+} // namespace literals
+} // namespace simdjson
+
 #endif // SIMDJSON_STATIC_REFLECTION
 #endif // SIMDJSON_GENERIC_COMPILE_TIME_JSON_H
 /* end file simdjson/compile_time_json.h */
@@ -186748,42 +239591,649 @@ consteval auto operator ""_json() {
 #if SIMDJSON_STATIC_REFLECTION

 /* skipped duplicate #include "simdjson/compile_time_json.h" */
-#include <array>
-#include <cstdint>
-#include <meta>
-#include <string_view>
+/* including simdjson/internal/fast_float.h: #include "simdjson/internal/fast_float.h" */
+/* begin file simdjson/internal/fast_float.h */
+// Vendored from fast_float v8.2.10, generated by tools/vendor_fast_float.sh.
+// Do not edit by hand; re-run the script to update.
+//
+//   https://github.com/fastfloat/fast_float
+//   Licensed under Apache-2.0 OR MIT OR BSL-1.0, at your option.
+//
+// simdjson uses this for two things that its own number parser cannot do:
+//   * the slow path for numbers with more than 19 significant digits, where
+//     fast_float's bigint comparison is several times quicker than the
+//     Wuffs-derived decimal shifting it replaced (see src/from_chars.cpp), and
+//   * correctly rounded parsing inside a constant expression, which the runtime
+//     path cannot offer because it relies on memcpy and __uint128_t (see
+//     compile_time_json-inl.h).
+//
+// Two edits are applied by the script. Every fast_float name is rewritten so
+// that this copy cannot collide with a copy of fast_float that the surrounding
+// program includes for itself: namespace fast_float -> simdjson_fast_float,
+// FASTFLOAT_* -> SIMDJSON_FASTFLOAT_*, fastfloat_* -> simdjson_fastfloat_*. And
+// the accented letters in the attribution comments below are folded to ASCII,
+// to keep the tree ASCII-only; no disrespect to the people named is intended.
+// simdjson_fast_float by Daniel Lemire
+// simdjson_fast_float by Joao Paulo Magalhaes
+//
+//
+// with contributions from Eugene Golushkov
+// with contributions from Maksim Kita
+// with contributions from Marcin Wojdyr
+// with contributions from Neal Richardson
+// with contributions from Tim Paine
+// with contributions from Fabio Pellacini
+// with contributions from Lenard Szolnoki
+// with contributions from Jan Pharago
+// with contributions from Maya Warrier
+// with contributions from Taha Khokhar
+// with contributions from Anders Dalvander
+//
+//
+// Licensed under the Apache License, Version 2.0, or the
+// MIT License or the Boost License. This file may not be copied,
+// modified, or distributed except according to those terms.
+//
+// MIT License Notice
+//
+//    MIT License
+//
+//    Copyright (c) 2021 The simdjson_fast_float authors
+//
+//    Permission is hereby granted, free of charge, to any
+//    person obtaining a copy of this software and associated
+//    documentation files (the "Software"), to deal in the
+//    Software without restriction, including without
+//    limitation the rights to use, copy, modify, merge,
+//    publish, distribute, sublicense, and/or sell copies of
+//    the Software, and to permit persons to whom the Software
+//    is furnished to do so, subject to the following
+//    conditions:
+//
+//    The above copyright notice and this permission notice
+//    shall be included in all copies or substantial portions
+//    of the Software.
+//
+//    THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF
+//    ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED
+//    TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A
+//    PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT
+//    SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY
+//    CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION
+//    OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR
+//    IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
+//    DEALINGS IN THE SOFTWARE.
+//
+// Apache License (Version 2.0) Notice
+//
+//    Copyright 2021 The simdjson_fast_float authors
+//    Licensed under the Apache License, Version 2.0 (the "License");
+//    you may not use this file except in compliance with the License.
+//    You may obtain a copy of the License at
+//
+//    http://www.apache.org/licenses/LICENSE-2.0
+//
+//    Unless required by applicable law or agreed to in writing, software
+//    distributed under the License is distributed on an "AS IS" BASIS,
+//    WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+//    See the License for the specific language governing permissions and
+//
+// BOOST License Notice
+//
+//    Boost Software License - Version 1.0 - August 17th, 2003
+//
+//    Permission is hereby granted, free of charge, to any person or organization
+//    obtaining a copy of the software and accompanying documentation covered by
+//    this license (the "Software") to use, reproduce, display, distribute,
+//    execute, and transmit the Software, and to prepare derivative works of the
+//    Software, and to permit third-parties to whom the Software is furnished to
+//    do so, all subject to the following:
+//
+//    The copyright notices in the Software and this entire statement, including
+//    the above license grant, this restriction and the following disclaimer,
+//    must be included in all copies of the Software, in whole or in part, and
+//    all derivative works of the Software, unless such copies or derivative
+//    works are solely in the form of machine-executable object code generated by
+//    a source language processor.
+//
+//    THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
+//    IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
+//    FITNESS FOR A PARTICULAR PURPOSE, TITLE AND NON-INFRINGEMENT. IN NO EVENT
+//    SHALL THE COPYRIGHT HOLDERS OR ANYONE DISTRIBUTING THE SOFTWARE BE LIABLE
+//    FOR ANY DAMAGES OR OTHER LIABILITY, WHETHER IN CONTRACT, TORT OR OTHERWISE,
+//    ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
+//    DEALINGS IN THE SOFTWARE.
+//

-#include <algorithm>
-#include <array>
-#include <charconv>
+#ifndef SIMDJSON_FASTFLOAT_CONSTEXPR_FEATURE_DETECT_H
+#define SIMDJSON_FASTFLOAT_CONSTEXPR_FEATURE_DETECT_H
+
+#ifdef __has_include
+#if __has_include(<version>)
+#include <version>
+#endif
+#endif
+
+// Testing for https://wg21.link/N3652, adopted in C++14
+#if defined(__cpp_constexpr) && __cpp_constexpr >= 201304
+#define SIMDJSON_FASTFLOAT_CONSTEXPR14 constexpr
+#else
+#define SIMDJSON_FASTFLOAT_CONSTEXPR14
+#endif
+
+#if defined(__cpp_lib_bit_cast) && __cpp_lib_bit_cast >= 201806L
+#define SIMDJSON_FASTFLOAT_HAS_BIT_CAST 1
+#else
+#define SIMDJSON_FASTFLOAT_HAS_BIT_CAST 0
+#endif
+
+#if defined(__cpp_lib_is_constant_evaluated) &&                                \
+    __cpp_lib_is_constant_evaluated >= 201811L
+#define SIMDJSON_FASTFLOAT_HAS_IS_CONSTANT_EVALUATED 1
+#else
+#define SIMDJSON_FASTFLOAT_HAS_IS_CONSTANT_EVALUATED 0
+#endif
+
+#if defined(__cpp_if_constexpr) && __cpp_if_constexpr >= 201606L
+#define SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(x) if constexpr (x)
+#else
+#define SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(x) if (x)
+#endif
+
+// Testing for relevant C++20 constexpr library features
+#if SIMDJSON_FASTFLOAT_HAS_IS_CONSTANT_EVALUATED && SIMDJSON_FASTFLOAT_HAS_BIT_CAST &&           \
+    defined(__cpp_lib_constexpr_algorithms) &&                                 \
+    __cpp_lib_constexpr_algorithms >= 201806L /*For std::copy and std::fill*/
+#define SIMDJSON_FASTFLOAT_CONSTEXPR20 constexpr
+#define SIMDJSON_FASTFLOAT_IS_CONSTEXPR 1
+#else
+#define SIMDJSON_FASTFLOAT_CONSTEXPR20
+#define SIMDJSON_FASTFLOAT_IS_CONSTEXPR 0
+#endif
+
+#if __cplusplus >= 201703L || (defined(_MSVC_LANG) && _MSVC_LANG >= 201703L)
+#define SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE 0
+#else
+#define SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE 1
+#endif
+
+#endif // SIMDJSON_FASTFLOAT_CONSTEXPR_FEATURE_DETECT_H
+
+#ifndef SIMDJSON_FASTFLOAT_FLOAT_COMMON_H
+#define SIMDJSON_FASTFLOAT_FLOAT_COMMON_H
+
+#include <cfloat>
+#include <cstddef>
 #include <cstdint>
-#include <expected>
-#include <meta>
-#include <string>
-#include <string_view>
-#include <vector>
+#include <cassert>
+#include <cstring>
+#include <limits>
+#include <type_traits>
+#include <system_error>
+#ifdef __has_include
+#if __has_include(<stdfloat>) && (__cplusplus > 202002L || (defined(_MSVC_LANG) && (_MSVC_LANG > 202002L)))
+#include <stdfloat>
+#endif
+#endif

-#define simdjson_consteval_error(...)                                          \
+#define SIMDJSON_FASTFLOAT_VERSION_MAJOR 8
+#define SIMDJSON_FASTFLOAT_VERSION_MINOR 2
+#define SIMDJSON_FASTFLOAT_VERSION_PATCH 10
+
+#define SIMDJSON_FASTFLOAT_STRINGIZE_IMPL(x) #x
+#define SIMDJSON_FASTFLOAT_STRINGIZE(x) SIMDJSON_FASTFLOAT_STRINGIZE_IMPL(x)
+
+#define SIMDJSON_FASTFLOAT_VERSION_STR                                                  \
+  SIMDJSON_FASTFLOAT_STRINGIZE(SIMDJSON_FASTFLOAT_VERSION_MAJOR)                                 \
+  "." SIMDJSON_FASTFLOAT_STRINGIZE(SIMDJSON_FASTFLOAT_VERSION_MINOR) "." SIMDJSON_FASTFLOAT_STRINGIZE(    \
+      SIMDJSON_FASTFLOAT_VERSION_PATCH)
+
+#define SIMDJSON_FASTFLOAT_VERSION                                                      \
+  (SIMDJSON_FASTFLOAT_VERSION_MAJOR * 10000 + SIMDJSON_FASTFLOAT_VERSION_MINOR * 100 +           \
+   SIMDJSON_FASTFLOAT_VERSION_PATCH)
+
+namespace simdjson_fast_float {
+
+enum class chars_format : uint64_t;
+
+namespace detail {
+constexpr chars_format basic_json_fmt = chars_format(1 << 5);
+constexpr chars_format basic_fortran_fmt = chars_format(1 << 6);
+} // namespace detail
+
+enum class chars_format : uint64_t {
+  scientific = 1 << 0,
+  fixed = 1 << 2,
+  hex = 1 << 3,
+  no_infnan = 1 << 4,
+  // RFC 8259: https://datatracker.ietf.org/doc/html/rfc8259#section-6
+  json = uint64_t(detail::basic_json_fmt) | fixed | scientific | no_infnan,
+  // Extension of RFC 8259 where, e.g., "inf" and "nan" are allowed.
+  json_or_infnan = uint64_t(detail::basic_json_fmt) | fixed | scientific,
+  fortran = uint64_t(detail::basic_fortran_fmt) | fixed | scientific,
+  general = fixed | scientific,
+  allow_leading_plus = 1 << 7,
+  skip_white_space = 1 << 8,
+};
+
+template <typename UC> struct from_chars_result_t {
+  UC const *ptr;
+  std::errc ec;
+
+  // https://www.open-std.org/jtc1/sc22/wg21/docs/papers/2023/p2497r0.html
+  constexpr explicit operator bool() const noexcept {
+    return ec == std::errc();
+  }
+};
+
+using from_chars_result = from_chars_result_t<char>;
+
+template <typename UC> struct parse_options_t {
+  constexpr explicit parse_options_t(chars_format fmt = chars_format::general,
+                                     UC dot = UC('.'), int b = 10)
+      : format(fmt), decimal_point(dot), base(b) {}
+
+  /** Which number formats are accepted */
+  chars_format format;
+  /** The character used as decimal point */
+  UC decimal_point;
+  /** The base used for integers */
+  int base;
+};
+
+using parse_options = parse_options_t<char>;
+
+} // namespace simdjson_fast_float
+
+#if SIMDJSON_FASTFLOAT_HAS_BIT_CAST
+#include <bit>
+#endif
+
+#if (defined(__x86_64) || defined(__x86_64__) || defined(_M_X64) ||            \
+     defined(__amd64) || defined(__aarch64__) || defined(_M_ARM64) ||          \
+     defined(__MINGW64__) || defined(__s390x__) ||                             \
+     (defined(__ppc64__) || defined(__PPC64__) || defined(__ppc64le__) ||      \
+      defined(__PPC64LE__)) ||                                                 \
+     defined(__loongarch64) || (defined(__riscv) && __riscv_xlen == 64))
+#define SIMDJSON_FASTFLOAT_64BIT 1
+#elif (defined(__i386) || defined(__i386__) || defined(_M_IX86) ||             \
+       defined(__arm__) || defined(_M_ARM) || defined(__ppc__) ||              \
+       defined(__MINGW32__) || defined(__EMSCRIPTEN__) ||                      \
+       (defined(__riscv) && __riscv_xlen == 32))
+#define SIMDJSON_FASTFLOAT_32BIT 1
+#else
+  // Need to check incrementally, since SIZE_MAX is a size_t, avoid overflow.
+// We can never tell the register width, but the SIZE_MAX is a good
+// approximation. UINTPTR_MAX and INTPTR_MAX are optional, so avoid them for max
+// portability.
+#if SIZE_MAX == 0xffff
+#error Unknown platform (16-bit, unsupported)
+#elif SIZE_MAX == 0xffffffff
+#define SIMDJSON_FASTFLOAT_32BIT 1
+#elif SIZE_MAX == 0xffffffffffffffff
+#define SIMDJSON_FASTFLOAT_64BIT 1
+#else
+#error Unknown platform (not 32-bit, not 64-bit?)
+#endif
+#endif
+
+#if ((defined(_WIN32) || defined(_WIN64)) && !defined(__clang__)) ||           \
+    (defined(_M_ARM64) && !defined(__MINGW32__))
+#include <intrin.h>
+#endif
+
+#if defined(_MSC_VER) && !defined(__clang__)
+#define SIMDJSON_FASTFLOAT_VISUAL_STUDIO 1
+#endif
+
+#if defined __BYTE_ORDER__ && defined __ORDER_BIG_ENDIAN__
+#define SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN (__BYTE_ORDER__ == __ORDER_BIG_ENDIAN__)
+#elif defined _WIN32
+#define SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN 0
+#else
+#if defined(__APPLE__) || defined(__FreeBSD__)
+#include <machine/endian.h>
+#elif defined(sun) || defined(__sun)
+#include <sys/byteorder.h>
+#elif defined(__MVS__)
+#include <sys/endian.h>
+#else
+#ifdef __has_include
+#if __has_include(<endian.h>)
+#include <endian.h>
+#endif //__has_include(<endian.h>)
+#endif //__has_include
+#endif
+#
+#ifndef __BYTE_ORDER__
+// safe choice
+#define SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN 0
+#endif
+#
+#ifndef __ORDER_LITTLE_ENDIAN__
+// safe choice
+#define SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN 0
+#endif
+#
+#if __BYTE_ORDER__ == __ORDER_LITTLE_ENDIAN__
+#define SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN 0
+#else
+#define SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN 1
+#endif
+#endif
+
+#if defined(__SSE2__) || (defined(SIMDJSON_FASTFLOAT_VISUAL_STUDIO) &&                  \
+                          (defined(_M_AMD64) || defined(_M_X64) ||             \
+                           (defined(_M_IX86_FP) && _M_IX86_FP == 2)))
+#define SIMDJSON_FASTFLOAT_SSE2 1
+#endif
+
+#if defined(__aarch64__) || defined(_M_ARM64)
+#define SIMDJSON_FASTFLOAT_NEON 1
+#endif
+
+#if defined(SIMDJSON_FASTFLOAT_SSE2) || defined(SIMDJSON_FASTFLOAT_NEON)
+#define SIMDJSON_FASTFLOAT_HAS_SIMD 1
+#endif
+
+#if defined(__GNUC__)
+// disable -Wcast-align=strict (GCC only)
+#define SIMDJSON_FASTFLOAT_SIMD_DISABLE_WARNINGS                                        \
+  _Pragma("GCC diagnostic push")                                               \
+      _Pragma("GCC diagnostic ignored \"-Wcast-align\"")
+#else
+#define SIMDJSON_FASTFLOAT_SIMD_DISABLE_WARNINGS
+#endif
+
+#if defined(__GNUC__)
+#define SIMDJSON_FASTFLOAT_SIMD_RESTORE_WARNINGS _Pragma("GCC diagnostic pop")
+#else
+#define SIMDJSON_FASTFLOAT_SIMD_RESTORE_WARNINGS
+#endif
+
+#ifdef SIMDJSON_FASTFLOAT_VISUAL_STUDIO
+#define simdjson_fastfloat_really_inline __forceinline
+#else
+#define simdjson_fastfloat_really_inline inline __attribute__((always_inline))
+#endif
+
+// Branch-probability hint marking the rare slow-path branches as cold, so the
+// optimizer keeps the out-of-line slow-path re-parse off the hot path (and does
+// not duplicate the force-inlined hot scanner into the caller, which bloated
+// the hot frame and hurt ILP on some targets). Used at the call site as
+//   if simdjson_fastfloat_unlikely(cond) { ... }
+// (the macro supplies the parentheses). It expands to the standard [[unlikely]]
+// attribute when supported, otherwise to __builtin_expect on GCC/Clang, or
+// to a no-op elsewhere (e.g. pre-C++20 MSVC, which has no equivalent hint).
+#ifdef __has_cpp_attribute
+#if __has_cpp_attribute(unlikely) >= 201803L
+// g++-9 hits hits this branch, but then fails to compile
+// [[unlikely]]. This happens only with g++-9.
+#if !defined(__GNUC__) || (__GNUC__ != 9)
+#define SIMDJSON_FASTFLOAT_USE_UNLIKELY_ATTR
+#endif
+#endif
+#endif
+
+#ifdef SIMDJSON_FASTFLOAT_USE_UNLIKELY_ATTR
+#define simdjson_fastfloat_unlikely(x) (x) [[unlikely]]
+#elif defined(__GNUC__) || defined(__clang__)
+#define simdjson_fastfloat_unlikely(x) (__builtin_expect(!!(x), 0))
+#else
+#define simdjson_fastfloat_unlikely(x) (x)
+#endif
+
+#ifndef SIMDJSON_FASTFLOAT_ASSERT
+#define SIMDJSON_FASTFLOAT_ASSERT(x)                                                    \
   {                                                                            \
-    std::abort();                                                              \
+    static_cast<void>(x);                                                      \
+  }
+#endif
+
+#ifndef SIMDJSON_FASTFLOAT_DEBUG_ASSERT
+#define SIMDJSON_FASTFLOAT_DEBUG_ASSERT(x)                                              \
+  {                                                                            \
+    static_cast<void>(x);                                                      \
   }
+#endif

-namespace simdjson {
-namespace compile_time {
+// rust style `try!()` macro, or `?` operator
+#define SIMDJSON_FASTFLOAT_TRY(x)                                                       \
+  {                                                                            \
+    if (!(x))                                                                  \
+      return false;                                                            \
+  }

-/**
- * Namespace for number parsing utilities.
- * We seek to provide exact compile-time number parsing functions.
- * That is not trivial, but thankfully we can reuse much of the existing
- * simdjson functionality.
- * Importantly, it is not a trivial matter to provide correct rounding
- * for floating-point numbers at compile-time. The fast_float library
- * does it well.
- */
-namespace number_parsing {
+#define SIMDJSON_FASTFLOAT_ENABLE_IF(...)                                               \
+  typename std::enable_if<(__VA_ARGS__), int>::type
+
+namespace simdjson_fast_float {
+
+simdjson_fastfloat_really_inline constexpr bool cpp20_and_in_constexpr() {
+#if SIMDJSON_FASTFLOAT_HAS_IS_CONSTANT_EVALUATED
+  return std::is_constant_evaluated();
+#else
+  return false;
+#endif
+}
+
+template <typename T>
+struct is_supported_float_type
+    : std::integral_constant<
+          bool, std::is_same<T, double>::value || std::is_same<T, float>::value
+#ifdef __STDCPP_FLOAT64_T__
+                    || std::is_same<T, std::float64_t>::value
+#endif
+#ifdef __STDCPP_FLOAT32_T__
+                    || std::is_same<T, std::float32_t>::value
+#endif
+#ifdef __STDCPP_FLOAT16_T__
+                    || std::is_same<T, std::float16_t>::value
+#endif
+#ifdef __STDCPP_BFLOAT16_T__
+                    || std::is_same<T, std::bfloat16_t>::value
+#endif
+          > {
+};
+
+template <typename T>
+using equiv_uint_t = typename std::conditional<
+    sizeof(T) == 1, uint8_t,
+    typename std::conditional<
+        sizeof(T) == 2, uint16_t,
+        typename std::conditional<sizeof(T) == 4, uint32_t,
+                                  uint64_t>::type>::type>::type;
+
+template <typename T> struct is_supported_integer_type : std::is_integral<T> {};
+
+template <typename UC>
+struct is_supported_char_type
+    : std::integral_constant<bool, std::is_same<UC, char>::value ||
+                                       std::is_same<UC, wchar_t>::value ||
+                                       std::is_same<UC, char16_t>::value ||
+                                       std::is_same<UC, char32_t>::value
+#ifdef __cpp_char8_t
+                                       || std::is_same<UC, char8_t>::value
+#endif
+                             > {
+};
+
+template <typename UC>
+inline SIMDJSON_FASTFLOAT_CONSTEXPR14 bool
+simdjson_fastfloat_strncasecmp3(UC const *actual_mixedcase,
+                       UC const *expected_lowercase) {
+  uint64_t mask{0};
+  SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 1) { mask = 0x2020202020202020; }
+  else SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 2) {
+    mask = 0x0020002000200020;
+  }
+  else SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 4) {
+    mask = 0x0000002000000020;
+  }
+  else {
+    return false;
+  }
+
+  uint64_t val1{0}, val2{0};
+  if (cpp20_and_in_constexpr()) {
+    for (size_t i = 0; i < 3; i++) {
+      if ((actual_mixedcase[i] | 32) != expected_lowercase[i]) {
+        return false;
+      }
+    }
+    return true;
+  } else {
+    SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 1 || sizeof(UC) == 2) {
+      ::memcpy(&val1, actual_mixedcase, 3 * sizeof(UC));
+      ::memcpy(&val2, expected_lowercase, 3 * sizeof(UC));
+      val1 |= mask;
+      val2 |= mask;
+      return val1 == val2;
+    }
+    else SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 4) {
+      ::memcpy(&val1, actual_mixedcase, 2 * sizeof(UC));
+      ::memcpy(&val2, expected_lowercase, 2 * sizeof(UC));
+      val1 |= mask;
+      if (val1 != val2) {
+        return false;
+      }
+      return (actual_mixedcase[2] | 32) == (expected_lowercase[2]);
+    }
+    else {
+      return false;
+    }
+  }
+}
+
+template <typename UC>
+inline SIMDJSON_FASTFLOAT_CONSTEXPR14 bool
+simdjson_fastfloat_strncasecmp5(UC const *actual_mixedcase,
+                       UC const *expected_lowercase) {
+  uint64_t mask{0};
+  uint64_t val1{0}, val2{0};
+  if (cpp20_and_in_constexpr()) {
+    for (size_t i = 0; i < 5; i++) {
+      if ((actual_mixedcase[i] | 32) != expected_lowercase[i]) {
+        return false;
+      }
+    }
+    return true;
+  } else {
+    SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 1) {
+      mask = 0x2020202020202020;
+      ::memcpy(&val1, actual_mixedcase, 5 * sizeof(UC));
+      ::memcpy(&val2, expected_lowercase, 5 * sizeof(UC));
+      val1 |= mask;
+      val2 |= mask;
+      return val1 == val2;
+    }
+    else SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 2) {
+      mask = 0x0020002000200020;
+      ::memcpy(&val1, actual_mixedcase, 4 * sizeof(UC));
+      ::memcpy(&val2, expected_lowercase, 4 * sizeof(UC));
+      val1 |= mask;
+      if (val1 != val2) {
+        return false;
+      }
+      return (actual_mixedcase[4] | 32) == (expected_lowercase[4]);
+    }
+    else SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 4) {
+      mask = 0x0000002000000020;
+      ::memcpy(&val1, actual_mixedcase, 2 * sizeof(UC));
+      ::memcpy(&val2, expected_lowercase, 2 * sizeof(UC));
+      val1 |= mask;
+      if (val1 != val2) {
+        return false;
+      }
+      ::memcpy(&val1, actual_mixedcase + 2, 2 * sizeof(UC));
+      ::memcpy(&val2, expected_lowercase + 2, 2 * sizeof(UC));
+      val1 |= mask;
+      if (val1 != val2) {
+        return false;
+      }
+      return (actual_mixedcase[4] | 32) == (expected_lowercase[4]);
+    }
+    else {
+      return false;
+    }
+  }
+}
+
+// Compares two ASCII strings in a case insensitive manner.
+template <typename UC>
+inline SIMDJSON_FASTFLOAT_CONSTEXPR14 bool
+simdjson_fastfloat_strncasecmp(UC const *actual_mixedcase, UC const *expected_lowercase,
+                      size_t length) {
+  uint64_t mask{0};
+  SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 1) { mask = 0x2020202020202020; }
+  else SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 2) {
+    mask = 0x0020002000200020;
+  }
+  else SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(sizeof(UC) == 4) {
+    mask = 0x0000002000000020;
+  }
+  else {
+    return false;
+  }
+
+  if (cpp20_and_in_constexpr()) {
+    for (size_t i = 0; i < length; i++) {
+      if ((actual_mixedcase[i] | 32) != expected_lowercase[i]) {
+        return false;
+      }
+    }
+    return true;
+  } else {
+    uint64_t val1{0}, val2{0};
+    size_t sz{8 / (sizeof(UC))};
+    for (size_t i = 0; i < length; i += sz) {
+      val1 = val2 = 0;
+      sz = sz < (length - i) ? sz : length - i;
+      ::memcpy(&val1, actual_mixedcase + i, sz * sizeof(UC));
+      ::memcpy(&val2, expected_lowercase + i, sz * sizeof(UC));
+      val1 |= mask;
+      val2 |= mask;
+      if (val1 != val2) {
+        return false;
+      }
+    }
+    return true;
+  }
+}
+
+#ifndef FLT_EVAL_METHOD
+#error "FLT_EVAL_METHOD should be defined, please include cfloat."
+#endif
+
+// a pointer and a length to a contiguous block of memory
+template <typename T> struct span {
+  T const *ptr;
+  size_t length;
+
+  constexpr span(T const *_ptr, size_t _length) : ptr(_ptr), length(_length) {}
+
+  constexpr span() : ptr(nullptr), length(0) {}
+
+  constexpr size_t len() const noexcept { return length; }
+
+  SIMDJSON_FASTFLOAT_CONSTEXPR14 const T &operator[](size_t index) const noexcept {
+    SIMDJSON_FASTFLOAT_DEBUG_ASSERT(index < length);
+    return ptr[index];
+  }
+};
+
+struct value128 {
+  uint64_t low;
+  uint64_t high;
+
+  constexpr value128(uint64_t _low, uint64_t _high) : low(_low), high(_high) {}
+
+  constexpr value128() : low(0), high(0) {}
+};

-// Counts the number of leading zeros in a 64-bit integer.
-consteval int leading_zeroes(uint64_t input_num, int last_bit = 0) {
+/* Helper C++14 constexpr generic implementation of leading_zeroes */
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 int
+leading_zeroes_generic(uint64_t input_num, int last_bit = 0) {
   if (input_num & uint64_t(0xffffffff00000000)) {
     input_num >>= 32;
     last_bit |= 32;
@@ -186809,199 +240259,4540 @@ consteval int leading_zeroes(uint64_t input_num, int last_bit = 0) {
   }
   return 63 - last_bit;
 }
-// Multiplies two 32-bit unsigned integers and returns a 64-bit result.
-consteval uint64_t emulu(uint32_t x, uint32_t y) { return x * (uint64_t)y; }
-consteval uint64_t umul128_generic(uint64_t ab, uint64_t cd, uint64_t *hi) {
-  uint64_t ad = emulu((uint32_t)(ab >> 32), (uint32_t)cd);
-  uint64_t bd = emulu((uint32_t)ab, (uint32_t)cd);
-  uint64_t adbc = ad + emulu((uint32_t)ab, (uint32_t)(cd >> 32));
-  uint64_t adbc_carry = (uint64_t)(adbc < ad);
+
+/* result might be undefined when input_num is zero */
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 int
+leading_zeroes(uint64_t input_num) {
+  assert(input_num > 0);
+  if (cpp20_and_in_constexpr()) {
+    return leading_zeroes_generic(input_num);
+  }
+#ifdef SIMDJSON_FASTFLOAT_VISUAL_STUDIO
+#if defined(_M_X64) || defined(_M_ARM64)
+  unsigned long leading_zero = 0;
+  // Search the mask data from most significant bit (MSB)
+  // to least significant bit (LSB) for a set bit (1).
+  _BitScanReverse64(&leading_zero, input_num);
+  return static_cast<int>(63 - leading_zero);
+#else
+  return leading_zeroes_generic(input_num);
+#endif
+#else
+  return __builtin_clzll(input_num);
+#endif
+}
+
+/* Helper C++14 constexpr generic implementation of countr_zero for 32-bit */
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 int
+countr_zero_generic_32(uint32_t input_num) {
+  if (input_num == 0) {
+    return 32;
+  }
+  int last_bit = 0;
+  if (!(input_num & 0x0000FFFF)) {
+    input_num >>= 16;
+    last_bit |= 16;
+  }
+  if (!(input_num & 0x00FF)) {
+    input_num >>= 8;
+    last_bit |= 8;
+  }
+  if (!(input_num & 0x0F)) {
+    input_num >>= 4;
+    last_bit |= 4;
+  }
+  if (!(input_num & 0x3)) {
+    input_num >>= 2;
+    last_bit |= 2;
+  }
+  if (!(input_num & 0x1)) {
+    last_bit |= 1;
+  }
+  return last_bit;
+}
+
+/* count trailing zeroes for 32-bit integers */
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 int
+countr_zero_32(uint32_t input_num) {
+  if (cpp20_and_in_constexpr()) {
+    return countr_zero_generic_32(input_num);
+  }
+#ifdef SIMDJSON_FASTFLOAT_VISUAL_STUDIO
+  unsigned long trailing_zero = 0;
+  if (_BitScanForward(&trailing_zero, input_num)) {
+    return static_cast<int>(trailing_zero);
+  }
+  return 32;
+#else
+  return input_num == 0 ? 32 : __builtin_ctz(input_num);
+#endif
+}
+
+// slow emulation routine for 32-bit
+simdjson_fastfloat_really_inline constexpr uint64_t emulu(uint32_t x, uint32_t y) {
+  return x * static_cast<uint64_t>(y);
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 uint64_t
+umul128_generic(uint64_t ab, uint64_t cd, uint64_t *hi) {
+  uint64_t ad =
+      emulu(static_cast<uint32_t>(ab >> 32), static_cast<uint32_t>(cd));
+  uint64_t bd = emulu(static_cast<uint32_t>(ab), static_cast<uint32_t>(cd));
+  uint64_t adbc =
+      ad + emulu(static_cast<uint32_t>(ab), static_cast<uint32_t>(cd >> 32));
+  uint64_t adbc_carry = static_cast<uint64_t>(adbc < ad);
   uint64_t lo = bd + (adbc << 32);
-  *hi = emulu((uint32_t)(ab >> 32), (uint32_t)(cd >> 32)) + (adbc >> 32) +
-        (adbc_carry << 32) + (uint64_t)(lo < bd);
+  *hi =
+      emulu(static_cast<uint32_t>(ab >> 32), static_cast<uint32_t>(cd >> 32)) +
+      (adbc >> 32) + (adbc_carry << 32) + static_cast<uint64_t>(lo < bd);
   return lo;
 }

-// Represents a 128-bit unsigned integer as two 64-bit parts.
-// We have a value128 struct elsewhere in the simdjson, but we
-// use a separate one here for clarity.
-struct value128 {
-  uint64_t low;
-  uint64_t high;
+#ifdef SIMDJSON_FASTFLOAT_32BIT

-  constexpr value128(uint64_t _low, uint64_t _high) : low(_low), high(_high) {}
-  constexpr value128() : low(0), high(0) {}
-};
+// slow emulation routine for 32-bit
+#if !defined(__MINGW64__)
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 uint64_t _umul128(uint64_t ab,
+                                                                uint64_t cd,
+                                                                uint64_t *hi) {
+  return umul128_generic(ab, cd, hi);
+}
+#endif // !__MINGW64__
+
+#endif // SIMDJSON_FASTFLOAT_32BIT

-// Multiplies two 64-bit integers and returns a 128-bit result as value128.
-consteval value128 full_multiplication(uint64_t a, uint64_t b) {
+// compute 64-bit a*b
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 value128
+full_multiplication(uint64_t a, uint64_t b) {
+  if (cpp20_and_in_constexpr()) {
+    value128 answer;
+    answer.low = umul128_generic(a, b, &answer.high);
+    return answer;
+  }
   value128 answer;
+#if defined(_M_ARM64) && !defined(__MINGW32__)
+  // ARM64 has native support for 64-bit multiplications, no need to emulate
+  // But MinGW on ARM64 doesn't have native support for 64-bit multiplications
+  answer.high = __umulh(a, b);
+  answer.low = a * b;
+#elif defined(SIMDJSON_FASTFLOAT_32BIT) || (defined(_WIN64) && !defined(__clang__) &&   \
+                                   !defined(_M_ARM64) && !defined(__GNUC__))
+  answer.low = _umul128(a, b, &answer.high); // _umul128 not available on ARM64
+#elif defined(SIMDJSON_FASTFLOAT_64BIT) && defined(__SIZEOF_INT128__)
+  __uint128_t r = static_cast<__uint128_t>(a) * b;
+  answer.low = uint64_t(r);
+  answer.high = uint64_t(r >> 64);
+#else
   answer.low = umul128_generic(a, b, &answer.high);
+#endif
   return answer;
 }

-// Converts mantissa and exponent to a double, considering the sign.
-consteval double to_double(uint64_t mantissa, int64_t exponent, bool negative) {
-  uint64_t sign_bit = negative ? (1ULL << 63) : 0;
-  uint64_t exponent_bits = (uint64_t(exponent) & 0x7FF) << 52;
-  uint64_t bits = sign_bit | exponent_bits | (mantissa & ((1ULL << 52) - 1));
-  return std::bit_cast<double>(bits);
+struct adjusted_mantissa {
+  uint64_t mantissa{0};
+  int32_t power2{0}; // a negative value indicates an invalid result
+  adjusted_mantissa() = default;
+
+  constexpr bool operator==(adjusted_mantissa const &o) const {
+    return mantissa == o.mantissa && power2 == o.power2;
+  }
+
+  constexpr bool operator!=(adjusted_mantissa const &o) const {
+    return mantissa != o.mantissa || power2 != o.power2;
+  }
+};
+
+// Bias so we can get the real exponent with an invalid adjusted_mantissa.
+constexpr static int32_t invalid_am_bias = -0x8000;
+
+// used for binary_format_lookup_tables<T>::max_mantissa
+constexpr uint64_t constant_55555 = 5 * 5 * 5 * 5 * 5;
+
+template <typename T, typename U = void> struct binary_format_lookup_tables;
+
+template <typename T> struct binary_format : binary_format_lookup_tables<T> {
+  using equiv_uint = equiv_uint_t<T>;
+
+  static constexpr int mantissa_explicit_bits();
+  static constexpr int minimum_exponent();
+  static constexpr int infinite_power();
+  static constexpr int sign_index();
+  static constexpr int
+  min_exponent_fast_path(); // used when fegetround() == FE_TONEAREST
+  static constexpr int max_exponent_fast_path();
+  static constexpr int max_exponent_round_to_even();
+  static constexpr int min_exponent_round_to_even();
+  static constexpr uint64_t max_mantissa_fast_path(int64_t power);
+  static constexpr uint64_t
+  max_mantissa_fast_path(); // used when fegetround() == FE_TONEAREST
+  static constexpr int largest_power_of_ten();
+  static constexpr int smallest_power_of_ten();
+  static constexpr T exact_power_of_ten(int64_t power);
+  static constexpr size_t max_digits();
+  static constexpr equiv_uint exponent_mask();
+  static constexpr equiv_uint mantissa_mask();
+  static constexpr equiv_uint hidden_bit_mask();
+};
+
+template <typename U> struct binary_format_lookup_tables<double, U> {
+  static constexpr double powers_of_ten[] = {
+      1e0,  1e1,  1e2,  1e3,  1e4,  1e5,  1e6,  1e7,  1e8,  1e9,  1e10, 1e11,
+      1e12, 1e13, 1e14, 1e15, 1e16, 1e17, 1e18, 1e19, 1e20, 1e21, 1e22};
+
+  // Largest integer value v so that (5**index * v) <= 1<<53.
+  // 0x20000000000000 == 1 << 53
+  static constexpr uint64_t max_mantissa[] = {
+      0x20000000000000,
+      0x20000000000000 / 5,
+      0x20000000000000 / (5 * 5),
+      0x20000000000000 / (5 * 5 * 5),
+      0x20000000000000 / (5 * 5 * 5 * 5),
+      0x20000000000000 / (constant_55555),
+      0x20000000000000 / (constant_55555 * 5),
+      0x20000000000000 / (constant_55555 * 5 * 5),
+      0x20000000000000 / (constant_55555 * 5 * 5 * 5),
+      0x20000000000000 / (constant_55555 * 5 * 5 * 5 * 5),
+      0x20000000000000 / (constant_55555 * constant_55555),
+      0x20000000000000 / (constant_55555 * constant_55555 * 5),
+      0x20000000000000 / (constant_55555 * constant_55555 * 5 * 5),
+      0x20000000000000 / (constant_55555 * constant_55555 * 5 * 5 * 5),
+      0x20000000000000 / (constant_55555 * constant_55555 * constant_55555),
+      0x20000000000000 / (constant_55555 * constant_55555 * constant_55555 * 5),
+      0x20000000000000 /
+          (constant_55555 * constant_55555 * constant_55555 * 5 * 5),
+      0x20000000000000 /
+          (constant_55555 * constant_55555 * constant_55555 * 5 * 5 * 5),
+      0x20000000000000 /
+          (constant_55555 * constant_55555 * constant_55555 * 5 * 5 * 5 * 5),
+      0x20000000000000 /
+          (constant_55555 * constant_55555 * constant_55555 * constant_55555),
+      0x20000000000000 / (constant_55555 * constant_55555 * constant_55555 *
+                          constant_55555 * 5),
+      0x20000000000000 / (constant_55555 * constant_55555 * constant_55555 *
+                          constant_55555 * 5 * 5),
+      0x20000000000000 / (constant_55555 * constant_55555 * constant_55555 *
+                          constant_55555 * 5 * 5 * 5),
+      0x20000000000000 / (constant_55555 * constant_55555 * constant_55555 *
+                          constant_55555 * 5 * 5 * 5 * 5)};
+};
+
+#if SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE
+
+template <typename U>
+constexpr double binary_format_lookup_tables<double, U>::powers_of_ten[];
+
+template <typename U>
+constexpr uint64_t binary_format_lookup_tables<double, U>::max_mantissa[];
+
+#endif
+
+template <typename U> struct binary_format_lookup_tables<float, U> {
+  static constexpr float powers_of_ten[] = {1e0f, 1e1f, 1e2f, 1e3f, 1e4f, 1e5f,
+                                            1e6f, 1e7f, 1e8f, 1e9f, 1e10f};
+
+  // Largest integer value v so that (5**index * v) <= 1<<24.
+  // 0x1000000 == 1<<24
+  static constexpr uint64_t max_mantissa[] = {
+      0x1000000,
+      0x1000000 / 5,
+      0x1000000 / (5 * 5),
+      0x1000000 / (5 * 5 * 5),
+      0x1000000 / (5 * 5 * 5 * 5),
+      0x1000000 / (constant_55555),
+      0x1000000 / (constant_55555 * 5),
+      0x1000000 / (constant_55555 * 5 * 5),
+      0x1000000 / (constant_55555 * 5 * 5 * 5),
+      0x1000000 / (constant_55555 * 5 * 5 * 5 * 5),
+      0x1000000 / (constant_55555 * constant_55555),
+      0x1000000 / (constant_55555 * constant_55555 * 5)};
+};
+
+#if SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE
+
+template <typename U>
+constexpr float binary_format_lookup_tables<float, U>::powers_of_ten[];
+
+template <typename U>
+constexpr uint64_t binary_format_lookup_tables<float, U>::max_mantissa[];
+
+#endif
+
+template <>
+inline constexpr int binary_format<double>::min_exponent_fast_path() {
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+  return 0;
+#else
+  return -22;
+#endif
+}
+
+template <>
+inline constexpr int binary_format<float>::min_exponent_fast_path() {
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+  return 0;
+#else
+  return -10;
+#endif
 }

-// Attempts to compute i * 10^(power) exactly; and if "negative" is
-// true, negate the result.
-// Returns true on success, false on failure.
-// Failure suggests and invalid input or out-of-range result.
-consteval bool compute_float_64(int64_t power, uint64_t i, bool negative,
-                                double &d) {
-  if (i == 0) {
-    d = negative ? -0.0 : 0.0;
+template <>
+inline constexpr int binary_format<double>::mantissa_explicit_bits() {
+  return 52;
+}
+
+template <>
+inline constexpr int binary_format<float>::mantissa_explicit_bits() {
+  return 23;
+}
+
+template <>
+inline constexpr int binary_format<double>::max_exponent_round_to_even() {
+  return 23;
+}
+
+template <>
+inline constexpr int binary_format<float>::max_exponent_round_to_even() {
+  return 10;
+}
+
+template <>
+inline constexpr int binary_format<double>::min_exponent_round_to_even() {
+  return -4;
+}
+
+template <>
+inline constexpr int binary_format<float>::min_exponent_round_to_even() {
+  return -17;
+}
+
+template <> inline constexpr int binary_format<double>::minimum_exponent() {
+  return -1023;
+}
+
+template <> inline constexpr int binary_format<float>::minimum_exponent() {
+  return -127;
+}
+
+template <> inline constexpr int binary_format<double>::infinite_power() {
+  return 0x7FF;
+}
+
+template <> inline constexpr int binary_format<float>::infinite_power() {
+  return 0xFF;
+}
+
+template <> inline constexpr int binary_format<double>::sign_index() {
+  return 63;
+}
+
+template <> inline constexpr int binary_format<float>::sign_index() {
+  return 31;
+}
+
+template <>
+inline constexpr int binary_format<double>::max_exponent_fast_path() {
+  return 22;
+}
+
+template <>
+inline constexpr int binary_format<float>::max_exponent_fast_path() {
+  return 10;
+}
+
+template <>
+inline constexpr uint64_t binary_format<double>::max_mantissa_fast_path() {
+  return uint64_t(2) << mantissa_explicit_bits();
+}
+
+template <>
+inline constexpr uint64_t binary_format<float>::max_mantissa_fast_path() {
+  return uint64_t(2) << mantissa_explicit_bits();
+}
+
+// credit: Jakub Jelinek
+#ifdef __STDCPP_FLOAT16_T__
+template <typename U> struct binary_format_lookup_tables<std::float16_t, U> {
+  static constexpr std::float16_t powers_of_ten[] = {1e0f16, 1e1f16, 1e2f16,
+                                                     1e3f16, 1e4f16};
+
+  // Largest integer value v so that (5**index * v) <= 1<<11.
+  // 0x800 == 1<<11
+  static constexpr uint64_t max_mantissa[] = {0x800,
+                                              0x800 / 5,
+                                              0x800 / (5 * 5),
+                                              0x800 / (5 * 5 * 5),
+                                              0x800 / (5 * 5 * 5 * 5),
+                                              0x800 / (constant_55555)};
+};
+
+#if SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE
+
+template <typename U>
+constexpr std::float16_t
+    binary_format_lookup_tables<std::float16_t, U>::powers_of_ten[];
+
+template <typename U>
+constexpr uint64_t
+    binary_format_lookup_tables<std::float16_t, U>::max_mantissa[];
+
+#endif
+
+template <>
+inline constexpr std::float16_t
+binary_format<std::float16_t>::exact_power_of_ten(int64_t power) {
+  // Work around clang bug https://godbolt.org/z/zedh7rrhc
+  return static_cast<void>(powers_of_ten[0]), powers_of_ten[power];
+}
+
+template <>
+inline constexpr binary_format<std::float16_t>::equiv_uint
+binary_format<std::float16_t>::exponent_mask() {
+  return 0x7C00;
+}
+
+template <>
+inline constexpr binary_format<std::float16_t>::equiv_uint
+binary_format<std::float16_t>::mantissa_mask() {
+  return 0x03FF;
+}
+
+template <>
+inline constexpr binary_format<std::float16_t>::equiv_uint
+binary_format<std::float16_t>::hidden_bit_mask() {
+  return 0x0400;
+}
+
+template <>
+inline constexpr int binary_format<std::float16_t>::max_exponent_fast_path() {
+  return 4;
+}
+
+template <>
+inline constexpr int binary_format<std::float16_t>::mantissa_explicit_bits() {
+  return 10;
+}
+
+template <>
+inline constexpr uint64_t
+binary_format<std::float16_t>::max_mantissa_fast_path() {
+  return uint64_t(2) << mantissa_explicit_bits();
+}
+
+template <>
+inline constexpr uint64_t
+binary_format<std::float16_t>::max_mantissa_fast_path(int64_t power) {
+  // caller is responsible to ensure that
+  // power >= 0 && power <= 4
+  //
+  // Work around clang bug https://godbolt.org/z/zedh7rrhc
+  return static_cast<void>(max_mantissa[0]), max_mantissa[power];
+}
+
+template <>
+inline constexpr int binary_format<std::float16_t>::min_exponent_fast_path() {
+  return 0;
+}
+
+template <>
+inline constexpr int
+binary_format<std::float16_t>::max_exponent_round_to_even() {
+  return 5;
+}
+
+template <>
+inline constexpr int
+binary_format<std::float16_t>::min_exponent_round_to_even() {
+  return -22;
+}
+
+template <>
+inline constexpr int binary_format<std::float16_t>::minimum_exponent() {
+  return -15;
+}
+
+template <>
+inline constexpr int binary_format<std::float16_t>::infinite_power() {
+  return 0x1F;
+}
+
+template <> inline constexpr int binary_format<std::float16_t>::sign_index() {
+  return 15;
+}
+
+template <>
+inline constexpr int binary_format<std::float16_t>::largest_power_of_ten() {
+  return 4;
+}
+
+template <>
+inline constexpr int binary_format<std::float16_t>::smallest_power_of_ten() {
+  return -27;
+}
+
+template <>
+inline constexpr size_t binary_format<std::float16_t>::max_digits() {
+  return 22;
+}
+#endif // __STDCPP_FLOAT16_T__
+
+// credit: Jakub Jelinek
+#ifdef __STDCPP_BFLOAT16_T__
+template <typename U> struct binary_format_lookup_tables<std::bfloat16_t, U> {
+  static constexpr std::bfloat16_t powers_of_ten[] = {1e0bf16, 1e1bf16, 1e2bf16,
+                                                      1e3bf16};
+
+  // Largest integer value v so that (5**index * v) <= 1<<8.
+  // 0x100 == 1<<8
+  static constexpr uint64_t max_mantissa[] = {0x100, 0x100 / 5, 0x100 / (5 * 5),
+                                              0x100 / (5 * 5 * 5),
+                                              0x100 / (5 * 5 * 5 * 5)};
+};
+
+#if SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE
+
+template <typename U>
+constexpr std::bfloat16_t
+    binary_format_lookup_tables<std::bfloat16_t, U>::powers_of_ten[];
+
+template <typename U>
+constexpr uint64_t
+    binary_format_lookup_tables<std::bfloat16_t, U>::max_mantissa[];
+
+#endif
+
+template <>
+inline constexpr std::bfloat16_t
+binary_format<std::bfloat16_t>::exact_power_of_ten(int64_t power) {
+  // Work around clang bug https://godbolt.org/z/zedh7rrhc
+  return static_cast<void>(powers_of_ten[0]), powers_of_ten[power];
+}
+
+template <>
+inline constexpr int binary_format<std::bfloat16_t>::max_exponent_fast_path() {
+  return 3;
+}
+
+template <>
+inline constexpr binary_format<std::bfloat16_t>::equiv_uint
+binary_format<std::bfloat16_t>::exponent_mask() {
+  return 0x7F80;
+}
+
+template <>
+inline constexpr binary_format<std::bfloat16_t>::equiv_uint
+binary_format<std::bfloat16_t>::mantissa_mask() {
+  return 0x007F;
+}
+
+template <>
+inline constexpr binary_format<std::bfloat16_t>::equiv_uint
+binary_format<std::bfloat16_t>::hidden_bit_mask() {
+  return 0x0080;
+}
+
+template <>
+inline constexpr int binary_format<std::bfloat16_t>::mantissa_explicit_bits() {
+  return 7;
+}
+
+template <>
+inline constexpr uint64_t
+binary_format<std::bfloat16_t>::max_mantissa_fast_path() {
+  return uint64_t(2) << mantissa_explicit_bits();
+}
+
+template <>
+inline constexpr uint64_t
+binary_format<std::bfloat16_t>::max_mantissa_fast_path(int64_t power) {
+  // caller is responsible to ensure that
+  // power >= 0 && power <= 3
+  //
+  // Work around clang bug https://godbolt.org/z/zedh7rrhc
+  return static_cast<void>(max_mantissa[0]), max_mantissa[power];
+}
+
+template <>
+inline constexpr int binary_format<std::bfloat16_t>::min_exponent_fast_path() {
+  return 0;
+}
+
+template <>
+inline constexpr int
+binary_format<std::bfloat16_t>::max_exponent_round_to_even() {
+  return 3;
+}
+
+template <>
+inline constexpr int
+binary_format<std::bfloat16_t>::min_exponent_round_to_even() {
+  return -24;
+}
+
+template <>
+inline constexpr int binary_format<std::bfloat16_t>::minimum_exponent() {
+  return -127;
+}
+
+template <>
+inline constexpr int binary_format<std::bfloat16_t>::infinite_power() {
+  return 0xFF;
+}
+
+template <> inline constexpr int binary_format<std::bfloat16_t>::sign_index() {
+  return 15;
+}
+
+template <>
+inline constexpr int binary_format<std::bfloat16_t>::largest_power_of_ten() {
+  return 38;
+}
+
+template <>
+inline constexpr int binary_format<std::bfloat16_t>::smallest_power_of_ten() {
+  return -60;
+}
+
+template <>
+inline constexpr size_t binary_format<std::bfloat16_t>::max_digits() {
+  return 98;
+}
+#endif // __STDCPP_BFLOAT16_T__
+
+template <>
+inline constexpr uint64_t
+binary_format<double>::max_mantissa_fast_path(int64_t power) {
+  // caller is responsible to ensure that
+  // power >= 0 && power <= 22
+  //
+  // Work around clang bug https://godbolt.org/z/zedh7rrhc
+  return static_cast<void>(max_mantissa[0]), max_mantissa[power];
+}
+
+template <>
+inline constexpr uint64_t
+binary_format<float>::max_mantissa_fast_path(int64_t power) {
+  // caller is responsible to ensure that
+  // power >= 0 && power <= 10
+  //
+  // Work around clang bug https://godbolt.org/z/zedh7rrhc
+  return static_cast<void>(max_mantissa[0]), max_mantissa[power];
+}
+
+template <>
+inline constexpr double
+binary_format<double>::exact_power_of_ten(int64_t power) {
+  // Work around clang bug https://godbolt.org/z/zedh7rrhc
+  return static_cast<void>(powers_of_ten[0]), powers_of_ten[power];
+}
+
+template <>
+inline constexpr float binary_format<float>::exact_power_of_ten(int64_t power) {
+  // Work around clang bug https://godbolt.org/z/zedh7rrhc
+  return static_cast<void>(powers_of_ten[0]), powers_of_ten[power];
+}
+
+template <> inline constexpr int binary_format<double>::largest_power_of_ten() {
+  return 308;
+}
+
+template <> inline constexpr int binary_format<float>::largest_power_of_ten() {
+  return 38;
+}
+
+template <>
+inline constexpr int binary_format<double>::smallest_power_of_ten() {
+  return -342;
+}
+
+template <> inline constexpr int binary_format<float>::smallest_power_of_ten() {
+  return -64;
+}
+
+template <> inline constexpr size_t binary_format<double>::max_digits() {
+  return 769;
+}
+
+template <> inline constexpr size_t binary_format<float>::max_digits() {
+  return 114;
+}
+
+template <>
+inline constexpr binary_format<float>::equiv_uint
+binary_format<float>::exponent_mask() {
+  return 0x7F800000;
+}
+
+template <>
+inline constexpr binary_format<double>::equiv_uint
+binary_format<double>::exponent_mask() {
+  return 0x7FF0000000000000;
+}
+
+template <>
+inline constexpr binary_format<float>::equiv_uint
+binary_format<float>::mantissa_mask() {
+  return 0x007FFFFF;
+}
+
+template <>
+inline constexpr binary_format<double>::equiv_uint
+binary_format<double>::mantissa_mask() {
+  return 0x000FFFFFFFFFFFFF;
+}
+
+template <>
+inline constexpr binary_format<float>::equiv_uint
+binary_format<float>::hidden_bit_mask() {
+  return 0x00800000;
+}
+
+template <>
+inline constexpr binary_format<double>::equiv_uint
+binary_format<double>::hidden_bit_mask() {
+  return 0x0010000000000000;
+}
+
+template <typename T>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 void
+to_float(bool negative, adjusted_mantissa am, T &value) {
+  using equiv_uint = equiv_uint_t<T>;
+  equiv_uint word = equiv_uint(am.mantissa);
+  word = equiv_uint(word | equiv_uint(am.power2)
+                               << binary_format<T>::mantissa_explicit_bits());
+  word =
+      equiv_uint(word | equiv_uint(negative) << binary_format<T>::sign_index());
+#if SIMDJSON_FASTFLOAT_HAS_BIT_CAST
+  value = std::bit_cast<T>(word);
+#else
+  ::memcpy(&value, &word, sizeof(T));
+#endif
+}
+
+template <typename = void> struct space_lut {
+  static constexpr bool value[] = {
+      0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+      0, 0, 0, 0, 0, 0, 0, 0, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+      0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+      0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+      0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+      0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+      0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+      0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+      0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+      0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
+      0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0};
+};
+
+#if SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE
+
+template <typename T> constexpr bool space_lut<T>::value[];
+
+#endif
+
+template <typename UC> constexpr bool is_space(UC c) {
+  // wchar_t and char can be signed, so a negative code unit slips past a plain
+  // `c < 256` and then indexes the table by its truncated low byte. Compare as
+  // unsigned, matching the care taken in ch_to_digit.
+  using UnsignedUC = typename std::make_unsigned<UC>::type;
+  return static_cast<UnsignedUC>(c) < 256 && space_lut<>::value[uint8_t(c)];
+}
+
+template <typename UC> static constexpr uint64_t int_cmp_zeros() {
+  static_assert((sizeof(UC) == 1) || (sizeof(UC) == 2) || (sizeof(UC) == 4),
+                "Unsupported character size");
+  return (sizeof(UC) == 1) ? 0x3030303030303030
+         : (sizeof(UC) == 2)
+             ? (uint64_t(UC('0')) << 48 | uint64_t(UC('0')) << 32 |
+                uint64_t(UC('0')) << 16 | UC('0'))
+             : (uint64_t(UC('0')) << 32 | UC('0'));
+}
+
+template <typename UC> static constexpr int int_cmp_len() {
+  return sizeof(uint64_t) / sizeof(UC);
+}
+
+template <typename UC> constexpr UC const *str_const_nan();
+
+template <> constexpr char const *str_const_nan<char>() { return "nan"; }
+
+template <> constexpr wchar_t const *str_const_nan<wchar_t>() { return L"nan"; }
+
+template <> constexpr char16_t const *str_const_nan<char16_t>() {
+  return u"nan";
+}
+
+template <> constexpr char32_t const *str_const_nan<char32_t>() {
+  return U"nan";
+}
+
+#ifdef __cpp_char8_t
+template <> constexpr char8_t const *str_const_nan<char8_t>() {
+  return u8"nan";
+}
+#endif
+
+template <typename UC> constexpr UC const *str_const_inf();
+
+template <> constexpr char const *str_const_inf<char>() { return "infinity"; }
+
+template <> constexpr wchar_t const *str_const_inf<wchar_t>() {
+  return L"infinity";
+}
+
+template <> constexpr char16_t const *str_const_inf<char16_t>() {
+  return u"infinity";
+}
+
+template <> constexpr char32_t const *str_const_inf<char32_t>() {
+  return U"infinity";
+}
+
+#ifdef __cpp_char8_t
+template <> constexpr char8_t const *str_const_inf<char8_t>() {
+  return u8"infinity";
+}
+#endif
+
+template <typename = void> struct int_luts {
+  static constexpr uint8_t chdigit[] = {
+      255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+      255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+      255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+      255, 255, 255, 0,   1,   2,   3,   4,   5,   6,   7,   8,   9,   255, 255,
+      255, 255, 255, 255, 255, 10,  11,  12,  13,  14,  15,  16,  17,  18,  19,
+      20,  21,  22,  23,  24,  25,  26,  27,  28,  29,  30,  31,  32,  33,  34,
+      35,  255, 255, 255, 255, 255, 255, 10,  11,  12,  13,  14,  15,  16,  17,
+      18,  19,  20,  21,  22,  23,  24,  25,  26,  27,  28,  29,  30,  31,  32,
+      33,  34,  35,  255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+      255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+      255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+      255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+      255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+      255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+      255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+      255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+      255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
+      255};
+
+  static constexpr size_t maxdigits_u64[] = {
+      64, 41, 32, 28, 25, 23, 22, 21, 20, 19, 18, 18, 17, 17, 16, 16, 16, 16,
+      15, 15, 15, 15, 14, 14, 14, 14, 14, 14, 14, 13, 13, 13, 13, 13, 13};
+
+  static constexpr uint64_t min_safe_u64[] = {
+      9223372036854775808ull,  12157665459056928801ull, 4611686018427387904,
+      7450580596923828125,     4738381338321616896,     3909821048582988049,
+      9223372036854775808ull,  12157665459056928801ull, 10000000000000000000ull,
+      5559917313492231481,     2218611106740436992,     8650415919381337933,
+      2177953337809371136,     6568408355712890625,     1152921504606846976,
+      2862423051509815793,     6746640616477458432,     15181127029874798299ull,
+      1638400000000000000,     3243919932521508681,     6221821273427820544,
+      11592836324538749809ull, 876488338465357824,      1490116119384765625,
+      2481152873203736576,     4052555153018976267,     6502111422497947648,
+      10260628712958602189ull, 15943230000000000000ull, 787662783788549761,
+      1152921504606846976,     1667889514952984961,     2386420683693101056,
+      3379220508056640625,     4738381338321616896};
+};
+
+#if SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE
+
+template <typename T> constexpr uint8_t int_luts<T>::chdigit[];
+
+template <typename T> constexpr size_t int_luts<T>::maxdigits_u64[];
+
+template <typename T> constexpr uint64_t int_luts<T>::min_safe_u64[];
+
+#endif
+
+template <typename UC>
+simdjson_fastfloat_really_inline constexpr uint8_t ch_to_digit(UC c) {
+  // wchar_t and char can be signed, so we need to be careful.
+  using UnsignedUC = typename std::make_unsigned<UC>::type;
+  return int_luts<>::chdigit[static_cast<unsigned char>(
+      static_cast<UnsignedUC>(c) &
+      static_cast<UnsignedUC>(
+          -((static_cast<UnsignedUC>(c) & ~0xFFull) == 0)))];
+}
+
+simdjson_fastfloat_really_inline constexpr size_t max_digits_u64(int base) {
+  return int_luts<>::maxdigits_u64[base - 2];
+}
+
+// If a u64 is exactly max_digits_u64() in length, this is
+// the value below which it has definitely overflowed.
+simdjson_fastfloat_really_inline constexpr uint64_t min_safe_u64(int base) {
+  return int_luts<>::min_safe_u64[base - 2];
+}
+
+static_assert(std::is_same<equiv_uint_t<double>, uint64_t>::value,
+              "equiv_uint should be uint64_t for double");
+static_assert(std::numeric_limits<double>::is_iec559,
+              "double must fulfill the requirements of IEC 559 (IEEE 754)");
+
+static_assert(std::is_same<equiv_uint_t<float>, uint32_t>::value,
+              "equiv_uint should be uint32_t for float");
+static_assert(std::numeric_limits<float>::is_iec559,
+              "float must fulfill the requirements of IEC 559 (IEEE 754)");
+
+#ifdef __STDCPP_FLOAT64_T__
+static_assert(std::is_same<equiv_uint_t<std::float64_t>, uint64_t>::value,
+              "equiv_uint should be uint64_t for std::float64_t");
+static_assert(
+    std::numeric_limits<std::float64_t>::is_iec559,
+    "std::float64_t must fulfill the requirements of IEC 559 (IEEE 754)");
+
+template <>
+struct binary_format<std::float64_t> : public binary_format<double> {};
+#endif // __STDCPP_FLOAT64_T__
+
+#ifdef __STDCPP_FLOAT32_T__
+static_assert(std::is_same<equiv_uint_t<std::float32_t>, uint32_t>::value,
+              "equiv_uint should be uint32_t for std::float32_t");
+static_assert(
+    std::numeric_limits<std::float32_t>::is_iec559,
+    "std::float32_t must fulfill the requirements of IEC 559 (IEEE 754)");
+
+template <>
+struct binary_format<std::float32_t> : public binary_format<float> {};
+#endif // __STDCPP_FLOAT32_T__
+
+#ifdef __STDCPP_FLOAT16_T__
+static_assert(
+    std::is_same<binary_format<std::float16_t>::equiv_uint, uint16_t>::value,
+    "equiv_uint should be uint16_t for std::float16_t");
+static_assert(
+    std::numeric_limits<std::float16_t>::is_iec559,
+    "std::float16_t must fulfill the requirements of IEC 559 (IEEE 754)");
+#endif // __STDCPP_FLOAT16_T__
+
+#ifdef __STDCPP_BFLOAT16_T__
+static_assert(
+    std::is_same<binary_format<std::bfloat16_t>::equiv_uint, uint16_t>::value,
+    "equiv_uint should be uint16_t for std::bfloat16_t");
+static_assert(
+    std::numeric_limits<std::bfloat16_t>::is_iec559,
+    "std::bfloat16_t must fulfill the requirements of IEC 559 (IEEE 754)");
+#endif // __STDCPP_BFLOAT16_T__
+
+constexpr chars_format operator~(chars_format rhs) noexcept {
+  using int_type = std::underlying_type<chars_format>::type;
+  return static_cast<chars_format>(~static_cast<int_type>(rhs));
+}
+
+constexpr chars_format operator&(chars_format lhs, chars_format rhs) noexcept {
+  using int_type = std::underlying_type<chars_format>::type;
+  return static_cast<chars_format>(static_cast<int_type>(lhs) &
+                                   static_cast<int_type>(rhs));
+}
+
+constexpr chars_format operator|(chars_format lhs, chars_format rhs) noexcept {
+  using int_type = std::underlying_type<chars_format>::type;
+  return static_cast<chars_format>(static_cast<int_type>(lhs) |
+                                   static_cast<int_type>(rhs));
+}
+
+constexpr chars_format operator^(chars_format lhs, chars_format rhs) noexcept {
+  using int_type = std::underlying_type<chars_format>::type;
+  return static_cast<chars_format>(static_cast<int_type>(lhs) ^
+                                   static_cast<int_type>(rhs));
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 chars_format &
+operator&=(chars_format &lhs, chars_format rhs) noexcept {
+  return lhs = (lhs & rhs);
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 chars_format &
+operator|=(chars_format &lhs, chars_format rhs) noexcept {
+  return lhs = (lhs | rhs);
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 chars_format &
+operator^=(chars_format &lhs, chars_format rhs) noexcept {
+  return lhs = (lhs ^ rhs);
+}
+
+namespace detail {
+// adjust for deprecated feature macros
+constexpr chars_format adjust_for_feature_macros(chars_format fmt) {
+  return fmt
+#ifdef SIMDJSON_FASTFLOAT_ALLOWS_LEADING_PLUS
+         | chars_format::allow_leading_plus
+#endif
+#ifdef SIMDJSON_FASTFLOAT_SKIP_WHITE_SPACE
+         | chars_format::skip_white_space
+#endif
+      ;
+}
+} // namespace detail
+} // namespace simdjson_fast_float
+
+#endif
+
+
+#ifndef SIMDJSON_FASTFLOAT_FAST_FLOAT_H
+#define SIMDJSON_FASTFLOAT_FAST_FLOAT_H
+
+
+namespace simdjson_fast_float {
+/**
+ * This function parses the character sequence [first,last) for a number. It
+ * parses floating-point numbers expecting a locale-independent format
+ * equivalent to what is used by std::strtod in the default ("C") locale. The
+ * resulting floating-point value is the closest floating-point values (using
+ * either float or double), using the "round to even" convention for values that
+ * would otherwise fall right in-between two values. That is, we provide exact
+ * parsing according to the IEEE standard.
+ *
+ * Given a successful parse, the pointer (`ptr`) in the returned value is set to
+ * point right after the parsed number, and the `value` referenced is set to the
+ * parsed value. In case of error, the returned `ec` contains a representative
+ * error, otherwise the default (`std::errc()`) value is stored.
+ *
+ * The implementation does not throw and does not allocate memory (e.g., with
+ * `new` or `malloc`).
+ *
+ * Like the C++17 standard, the `simdjson_fast_float::from_chars` functions take an
+ * optional last argument of the type `simdjson_fast_float::chars_format`. It is a bitset
+ * value: we check whether `fmt & simdjson_fast_float::chars_format::fixed` and `fmt &
+ * simdjson_fast_float::chars_format::scientific` are set to determine whether we allow
+ * the fixed point and scientific notation respectively. The default is
+ * `simdjson_fast_float::chars_format::general` which allows both `fixed` and
+ * `scientific`.
+ */
+template <typename T, typename UC = char,
+          typename = SIMDJSON_FASTFLOAT_ENABLE_IF(is_supported_float_type<T>::value)>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+from_chars(UC const *first, UC const *last, T &value,
+           chars_format fmt = chars_format::general) noexcept;
+
+/**
+ * Like from_chars, but accepts an `options` argument to govern number parsing.
+ * Both for floating-point types and integer types.
+ */
+template <typename T, typename UC = char>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+from_chars_advanced(UC const *first, UC const *last, T &value,
+                    parse_options_t<UC> options) noexcept;
+
+/**
+ * This function multiplies an integer number by a power of 10 and returns
+ * the result as a double precision floating-point value that is correctly
+ * rounded. The resulting floating-point value is the closest floating-point
+ * value, using the "round to nearest, tie to even" convention for values that
+ * would otherwise fall right in-between two values. That is, we provide exact
+ * conversion according to the IEEE standard.
+ *
+ * On overflow infinity is returned, on underflow 0 is returned.
+ *
+ * The implementation does not throw and does not allocate memory (e.g., with
+ * `new` or `malloc`).
+ */
+SIMDJSON_FASTFLOAT_CONSTEXPR20 inline double
+integer_times_pow10(uint64_t mantissa, int decimal_exponent) noexcept;
+SIMDJSON_FASTFLOAT_CONSTEXPR20 inline double
+integer_times_pow10(int64_t mantissa, int decimal_exponent) noexcept;
+
+/**
+ * This function is a template overload of `integer_times_pow10()`
+ * that returns a floating-point value of type `T` that is one of
+ * supported floating-point types (e.g. `double`, `float`).
+ */
+template <typename T>
+SIMDJSON_FASTFLOAT_CONSTEXPR20
+    typename std::enable_if<is_supported_float_type<T>::value, T>::type
+    integer_times_pow10(uint64_t mantissa, int decimal_exponent) noexcept;
+template <typename T>
+SIMDJSON_FASTFLOAT_CONSTEXPR20
+    typename std::enable_if<is_supported_float_type<T>::value, T>::type
+    integer_times_pow10(int64_t mantissa, int decimal_exponent) noexcept;
+
+/**
+ * from_chars for integer types.
+ */
+template <typename T, typename UC = char,
+          typename = SIMDJSON_FASTFLOAT_ENABLE_IF(is_supported_integer_type<T>::value)>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+from_chars(UC const *first, UC const *last, T &value, int base = 10) noexcept;
+
+} // namespace simdjson_fast_float
+
+#endif // SIMDJSON_FASTFLOAT_FAST_FLOAT_H
+
+#ifndef SIMDJSON_FASTFLOAT_ASCII_NUMBER_H
+#define SIMDJSON_FASTFLOAT_ASCII_NUMBER_H
+
+#include <cctype>
+#include <cstdint>
+#include <cstring>
+#include <iterator>
+#include <limits>
+#include <type_traits>
+
+
+#ifdef SIMDJSON_FASTFLOAT_SSE2
+#include <emmintrin.h>
+#endif
+
+#ifdef SIMDJSON_FASTFLOAT_NEON
+#include <arm_neon.h>
+#endif
+
+namespace simdjson_fast_float {
+
+template <typename UC> simdjson_fastfloat_really_inline constexpr bool has_simd_opt() {
+#ifdef SIMDJSON_FASTFLOAT_HAS_SIMD
+  return std::is_same<UC, char16_t>::value;
+#else
+  return false;
+#endif
+}
+
+// Next function can be micro-optimized, but compilers are entirely
+// able to optimize it well.
+template <typename UC>
+simdjson_fastfloat_really_inline constexpr bool is_integer(UC c) noexcept {
+  return static_cast<unsigned>(c - UC('0')) <= 9u;
+}
+
+simdjson_fastfloat_really_inline constexpr uint64_t byteswap(uint64_t val) {
+  return (val & 0xFF00000000000000) >> 56 | (val & 0x00FF000000000000) >> 40 |
+         (val & 0x0000FF0000000000) >> 24 | (val & 0x000000FF00000000) >> 8 |
+         (val & 0x00000000FF000000) << 8 | (val & 0x0000000000FF0000) << 24 |
+         (val & 0x000000000000FF00) << 40 | (val & 0x00000000000000FF) << 56;
+}
+
+simdjson_fastfloat_really_inline constexpr uint32_t byteswap_32(uint32_t val) {
+  return (val >> 24) | ((val >> 8) & 0x0000FF00u) | ((val << 8) & 0x00FF0000u) |
+         (val << 24);
+}
+
+// Read 8 UC into a u64. Truncates UC if not char.
+template <typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 uint64_t
+read8_to_u64(UC const *chars) {
+  if (cpp20_and_in_constexpr() || !std::is_same<UC, char>::value) {
+    uint64_t val = 0;
+    for (int i = 0; i < 8; ++i) {
+      val |= uint64_t(uint8_t(*chars)) << (i * 8);
+      ++chars;
+    }
+    return val;
+  }
+  uint64_t val;
+  ::memcpy(&val, chars, sizeof(uint64_t));
+#if SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN == 1
+  // Need to read as-if the number was in little-endian order.
+  val = byteswap(val);
+#endif
+  return val;
+}
+
+// Read 4 UC into a u32. Truncates UC if not char.
+template <typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 uint32_t
+read4_to_u32(UC const *chars) {
+  if (cpp20_and_in_constexpr() || !std::is_same<UC, char>::value) {
+    uint32_t val = 0;
+    for (int i = 0; i < 4; ++i) {
+      val |= uint32_t(uint8_t(*chars)) << (i * 8);
+      ++chars;
+    }
+    return val;
+  }
+  uint32_t val;
+  ::memcpy(&val, chars, sizeof(uint32_t));
+#if SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN == 1
+  val = byteswap_32(val);
+#endif
+  return val;
+}
+#ifdef SIMDJSON_FASTFLOAT_SSE2
+
+simdjson_fastfloat_really_inline uint64_t simd_read8_to_u64(__m128i const data) {
+  SIMDJSON_FASTFLOAT_SIMD_DISABLE_WARNINGS
+  __m128i const packed = _mm_packus_epi16(data, data);
+#ifdef SIMDJSON_FASTFLOAT_64BIT
+  return uint64_t(_mm_cvtsi128_si64(packed));
+#else
+  uint64_t value;
+  // Visual Studio + older versions of GCC don't support _mm_storeu_si64
+  _mm_storel_epi64(reinterpret_cast<__m128i *>(&value), packed);
+  return value;
+#endif
+  SIMDJSON_FASTFLOAT_SIMD_RESTORE_WARNINGS
+}
+
+simdjson_fastfloat_really_inline uint64_t simd_read8_to_u64(char16_t const *chars) {
+  SIMDJSON_FASTFLOAT_SIMD_DISABLE_WARNINGS
+  return simd_read8_to_u64(
+      _mm_loadu_si128(reinterpret_cast<__m128i const *>(chars)));
+  SIMDJSON_FASTFLOAT_SIMD_RESTORE_WARNINGS
+}
+
+#elif defined(SIMDJSON_FASTFLOAT_NEON)
+
+simdjson_fastfloat_really_inline uint64_t simd_read8_to_u64(uint16x8_t const data) {
+  SIMDJSON_FASTFLOAT_SIMD_DISABLE_WARNINGS
+  uint8x8_t utf8_packed = vmovn_u16(data);
+  return vget_lane_u64(vreinterpret_u64_u8(utf8_packed), 0);
+  SIMDJSON_FASTFLOAT_SIMD_RESTORE_WARNINGS
+}
+
+simdjson_fastfloat_really_inline uint64_t simd_read8_to_u64(char16_t const *chars) {
+  SIMDJSON_FASTFLOAT_SIMD_DISABLE_WARNINGS
+  return simd_read8_to_u64(
+      vld1q_u16(reinterpret_cast<uint16_t const *>(chars)));
+  SIMDJSON_FASTFLOAT_SIMD_RESTORE_WARNINGS
+}
+
+#endif // SIMDJSON_FASTFLOAT_SSE2
+
+// MSVC SFINAE is broken pre-VS2017
+#if defined(_MSC_VER) && _MSC_VER <= 1900
+template <typename UC>
+#else
+template <typename UC, SIMDJSON_FASTFLOAT_ENABLE_IF(!has_simd_opt<UC>()) = 0>
+#endif
+// dummy for compile
+uint64_t simd_read8_to_u64(UC const *) {
+  return 0;
+}
+
+// credit  @aqrit
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 uint32_t
+parse_eight_digits_unrolled(uint64_t val) {
+  uint64_t const mask = 0x000000FF000000FF;
+  uint64_t const mul1 = 0x000F424000000064; // 100 + (1000000ULL << 32)
+  uint64_t const mul2 = 0x0000271000000001; // 1 + (10000ULL << 32)
+  val -= 0x3030303030303030;
+  val = (val * 10) + (val >> 8); // val = (val * 2561) >> 8;
+  val = (((val & mask) * mul1) + (((val >> 16) & mask) * mul2)) >> 32;
+  return uint32_t(val);
+}
+
+// Call this if chars are definitely 8 digits.
+template <typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 uint32_t
+parse_eight_digits_unrolled(UC const *chars) noexcept {
+  if (cpp20_and_in_constexpr() || !has_simd_opt<UC>()) {
+    return parse_eight_digits_unrolled(read8_to_u64(chars)); // truncation okay
+  }
+  return parse_eight_digits_unrolled(simd_read8_to_u64(chars));
+}
+
+// credit @aqrit
+simdjson_fastfloat_really_inline constexpr bool
+is_made_of_eight_digits_fast(uint64_t val) noexcept {
+  return !((((val + 0x4646464646464646) | (val - 0x3030303030303030)) &
+            0x8080808080808080));
+}
+
+simdjson_fastfloat_really_inline constexpr bool
+is_made_of_four_digits_fast(uint32_t val) noexcept {
+  return !((((val + 0x46464646) | (val - 0x30303030)) & 0x80808080));
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 uint32_t
+parse_four_digits_unrolled(uint32_t val) noexcept {
+  val -= 0x30303030;
+  val = (val * 10) + (val >> 8);
+  return (((val & 0x00FF00FF) * 0x00640001) >> 16) & 0xFFFF;
+}
+
+#ifdef SIMDJSON_FASTFLOAT_HAS_SIMD
+
+// Call this if chars might not be 8 digits.
+// Using this style (instead of is_made_of_eight_digits_fast() then
+// parse_eight_digits_unrolled()) ensures we don't load SIMD registers twice.
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 bool
+simd_parse_if_eight_digits_unrolled(char16_t const *chars,
+                                    uint64_t &i) noexcept {
+  if (cpp20_and_in_constexpr()) {
+    return false;
+  }
+#ifdef SIMDJSON_FASTFLOAT_SSE2
+  SIMDJSON_FASTFLOAT_SIMD_DISABLE_WARNINGS
+  __m128i const data =
+      _mm_loadu_si128(reinterpret_cast<__m128i const *>(chars));
+
+  // (x - '0') <= 9
+  // http://0x80.pl/articles/simd-parsing-int-sequences.html
+  __m128i const t0 = _mm_add_epi16(data, _mm_set1_epi16(32720));
+  __m128i const t1 = _mm_cmpgt_epi16(t0, _mm_set1_epi16(-32759));
+
+  if (_mm_movemask_epi8(t1) == 0) {
+    i = i * 100000000 + parse_eight_digits_unrolled(simd_read8_to_u64(data));
     return true;
+  } else
+    return false;
+  SIMDJSON_FASTFLOAT_SIMD_RESTORE_WARNINGS
+#elif defined(SIMDJSON_FASTFLOAT_NEON)
+  SIMDJSON_FASTFLOAT_SIMD_DISABLE_WARNINGS
+  uint16x8_t const data = vld1q_u16(reinterpret_cast<uint16_t const *>(chars));
+
+  // (x - '0') <= 9
+  // http://0x80.pl/articles/simd-parsing-int-sequences.html
+  uint16x8_t const t0 = vsubq_u16(data, vmovq_n_u16('0'));
+  uint16x8_t const mask = vcltq_u16(t0, vmovq_n_u16('9' - '0' + 1));
+
+  if (vminvq_u16(mask) == 0xFFFF) {
+    i = i * 100000000 + parse_eight_digits_unrolled(simd_read8_to_u64(data));
+    return true;
+  } else
+    return false;
+  SIMDJSON_FASTFLOAT_SIMD_RESTORE_WARNINGS
+#else
+  static_cast<void>(chars);
+  static_cast<void>(i);
+  return false;
+#endif // SIMDJSON_FASTFLOAT_SSE2
+}
+
+#endif // SIMDJSON_FASTFLOAT_HAS_SIMD
+
+// MSVC SFINAE is broken pre-VS2017
+#if defined(_MSC_VER) && _MSC_VER <= 1900
+template <typename UC>
+#else
+template <typename UC, SIMDJSON_FASTFLOAT_ENABLE_IF(!has_simd_opt<UC>()) = 0>
+#endif
+// dummy for compile
+bool simd_parse_if_eight_digits_unrolled(UC const *, uint64_t &) {
+  return 0;
+}
+
+template <typename UC, SIMDJSON_FASTFLOAT_ENABLE_IF(!std::is_same<UC, char>::value) = 0>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 void
+loop_parse_if_eight_digits(UC const *&p, UC const *const pend, uint64_t &i) {
+  if (!has_simd_opt<UC>()) {
+    return;
   }
-  int64_t exponent = (((152170 + 65536) * power) >> 16) + 1024 + 63;
-  int lz = leading_zeroes(i);
-  i <<= lz;
-  const uint32_t index =
-      2 * uint32_t(power - simdjson::internal::smallest_power);
-  value128 firstproduct = full_multiplication(
-      i, simdjson::internal::powers_template<>::power_of_five_128[index]);
-  if ((firstproduct.high & 0x1FF) == 0x1FF) {
-    value128 secondproduct = full_multiplication(
-        i, simdjson::internal::powers_template<>::power_of_five_128[index + 1]);
+  while ((std::distance(p, pend) >= 8) &&
+         simd_parse_if_eight_digits_unrolled(
+             p, i)) { // in rare cases, this will overflow, but that's ok
+    p += 8;
+  }
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 void
+loop_parse_if_eight_digits(char const *&p, char const *const pend,
+                           uint64_t &i) {
+  // optimizes better than parse_if_eight_digits_unrolled() for UC = char.
+  while ((std::distance(p, pend) >= 8) &&
+         is_made_of_eight_digits_fast(read8_to_u64(p))) {
+    i = i * 100000000 +
+        parse_eight_digits_unrolled(read8_to_u64(
+            p)); // in rare cases, this will overflow, but that's ok
+    p += 8;
+  }
+  // Consume a remaining 4-7 digit run in a single SWAR step instead of
+  // byte-by-byte (reuses the existing 4-digit helpers). The parsed result is
+  // identical either way. Historically gated to clang because gcc regressed on
+  // short remainders, but that verdict predates the span-elision restructure;
+  // with the leaner hot path the 4-digit step now wins on gcc as well.
+  if ((pend - p) >= 4) {
+    uint32_t const val4 = read4_to_u32(p);
+    if (is_made_of_four_digits_fast(val4)) {
+      i = i * 10000 +
+          parse_four_digits_unrolled(val4); // may overflow, that's ok
+      p += 4;
+    }
+  }
+}
+
+enum class parse_error {
+  no_error,
+  // [JSON-only] The minus sign must be followed by an integer.
+  missing_integer_after_sign,
+  // A sign must be followed by an integer or dot.
+  missing_integer_or_dot_after_sign,
+  // [JSON-only] The integer part must not have leading zeros.
+  leading_zeros_in_integer_part,
+  // [JSON-only] The integer part must have at least one digit.
+  no_digits_in_integer_part,
+  // [JSON-only] If there is a decimal point, there must be digits in the
+  // fractional part.
+  no_digits_in_fractional_part,
+  // The mantissa must have at least one digit.
+  no_digits_in_mantissa,
+  // Scientific notation requires an exponential part.
+  missing_exponential_part,
+};
+
+template <typename UC> struct parsed_number_string_t {
+  int64_t exponent{0};
+  uint64_t mantissa{0};
+  UC const *lastmatch{nullptr};
+  bool negative{false};
+  bool valid{false};
+  bool too_many_digits{false};
+  // contains the range of the significant digits
+  span<UC const> integer{};  // non-nullable
+  span<UC const> fraction{}; // nullable
+  parse_error error{parse_error::no_error};
+};
+
+using byte_span = span<char const>;
+using parsed_number_string = parsed_number_string_t<char>;
+
+template <typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 parsed_number_string_t<UC>
+report_parse_error(UC const *p, parse_error error) {
+  parsed_number_string_t<UC> answer;
+  answer.valid = false;
+  answer.lastmatch = p;
+  answer.error = error;
+  return answer;
+}
+
+// Assuming that you use no more than 19 digits, this will
+// parse an ASCII string.
+//
+// store_spans is a *runtime* flag (not a template parameter, deliberately: a
+// template would create a second instantiation of this whole function and the
+// extra icache pressure wipes out the gain). When false, the integer/fraction
+// spans (read only by the rare digit_comp slow path) are not materialized,
+// which keeps the fat parsed_number_string_t off the hot path. The caller
+// re-parses with store_spans=true if the slow path is actually reached.
+template <bool basic_json_fmt, typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 parsed_number_string_t<UC>
+parse_number_string(UC const *p, UC const *pend, parse_options_t<UC> options,
+                    bool store_spans = true) noexcept {
+  chars_format const fmt = detail::adjust_for_feature_macros(options.format);
+  UC const decimal_point = options.decimal_point;
+
+  parsed_number_string_t<UC> answer;
+  answer.valid = false;
+  answer.too_many_digits = false;
+  // assume p < pend, so dereference without checks;
+  answer.negative = (*p == UC('-'));
+  // C++17 20.19.3.(7.1) explicitly forbids '+' sign here
+  if ((*p == UC('-')) || (uint64_t(fmt & chars_format::allow_leading_plus) &&
+                          !basic_json_fmt && *p == UC('+'))) {
+    ++p;
+    if (p == pend) {
+      return report_parse_error<UC>(
+          p, parse_error::missing_integer_or_dot_after_sign);
+    }
+    SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(basic_json_fmt) {
+      if (!is_integer(*p)) { // a sign must be followed by an integer
+        return report_parse_error<UC>(p,
+                                      parse_error::missing_integer_after_sign);
+      }
+    }
+    else {
+      if (!is_integer(*p) &&
+          (*p !=
+           decimal_point)) { // a sign must be followed by an integer or the dot
+        return report_parse_error<UC>(
+            p, parse_error::missing_integer_or_dot_after_sign);
+      }
+    }
+  }
+  UC const *const start_digits = p;
+
+  uint64_t i = 0; // an unsigned int avoids signed overflows (which are bad)
+
+  // Straight-line unroll of the integer-part scan: most integer parts are
+  // 1-5 digits, so peeling the first iterations eliminates the loop back-edge
+  // for the common case. Semantics are identical to the original `while` loop:
+  // i = 10*i + digit, advancing p.
+  if ((p != pend) && is_integer(*p)) {
+    i = uint64_t(*p - UC('0'));
+    ++p;
+    if ((p != pend) && is_integer(*p)) {
+      i = 10 * i + uint64_t(*p - UC('0'));
+      ++p;
+      if ((p != pend) && is_integer(*p)) {
+        i = 10 * i + uint64_t(*p - UC('0'));
+        ++p;
+        if ((p != pend) && is_integer(*p)) {
+          i = 10 * i + uint64_t(*p - UC('0'));
+          ++p;
+          if ((p != pend) && is_integer(*p)) {
+            i = 10 * i + uint64_t(*p - UC('0'));
+            ++p;
+            while ((p != pend) && is_integer(*p)) {
+              // a multiplication by 10 is cheaper than an arbitrary integer
+              // multiplication
+              i = 10 * i +
+                  uint64_t(*p - UC('0')); // might overflow, handled later
+              ++p;
+            }
+          }
+        }
+      }
+    }
+  }
+  UC const *const end_of_integer_part = p;
+  int64_t digit_count = int64_t(end_of_integer_part - start_digits);
+  if (store_spans) {
+    answer.integer = span<UC const>(start_digits, size_t(digit_count));
+  }
+  SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(basic_json_fmt) {
+    // at least 1 digit in integer part, without leading zeros
+    if (digit_count == 0) {
+      return report_parse_error<UC>(p, parse_error::no_digits_in_integer_part);
+    }
+    if ((start_digits[0] == UC('0') && digit_count > 1)) {
+      return report_parse_error<UC>(start_digits,
+                                    parse_error::leading_zeros_in_integer_part);
+    }
+  }
+
+  int64_t exponent = 0;
+  bool const has_decimal_point = (p != pend) && (*p == decimal_point);
+  if (has_decimal_point) {
+    ++p;
+    UC const *before = p;
+    // can occur at most twice without overflowing, but let it occur more, since
+    // for integers with many digits, digit parsing is the primary bottleneck.
+    loop_parse_if_eight_digits(p, pend, i);
+
+    while ((p != pend) && is_integer(*p)) {
+      uint8_t digit = uint8_t(*p - UC('0'));
+      ++p;
+      i = i * 10 + digit; // in rare cases, this will overflow, but that's ok
+    }
+    exponent = before - p;
+    if (store_spans) {
+      answer.fraction = span<UC const>(before, size_t(p - before));
+    }
+    digit_count -= exponent;
+  }
+  SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(basic_json_fmt) {
+    // at least 1 digit in fractional part
+    if (has_decimal_point && exponent == 0) {
+      return report_parse_error<UC>(p,
+                                    parse_error::no_digits_in_fractional_part);
+    }
+  }
+  else if (digit_count == 0) { // we must have encountered at least one integer!
+    return report_parse_error<UC>(p, parse_error::no_digits_in_mantissa);
+  }
+  int64_t exp_number = 0; // explicit exponential part
+  if ((uint64_t(fmt & chars_format::scientific) && (p != pend) &&
+       ((UC('e') == *p) || (UC('E') == *p))) ||
+      (uint64_t(fmt & detail::basic_fortran_fmt) && (p != pend) &&
+       ((UC('+') == *p) || (UC('-') == *p) || (UC('d') == *p) ||
+        (UC('D') == *p)))) {
+    UC const *location_of_e = p;
+    if ((UC('e') == *p) || (UC('E') == *p) || (UC('d') == *p) ||
+        (UC('D') == *p)) {
+      ++p;
+    }
+    bool neg_exp = false;
+    if ((p != pend) && (UC('-') == *p)) {
+      neg_exp = true;
+      ++p;
+    } else if ((p != pend) &&
+               (UC('+') ==
+                *p)) { // '+' on exponent is allowed by C++17 20.19.3.(7.1)
+      ++p;
+    }
+    if ((p == pend) || !is_integer(*p)) {
+      if (!uint64_t(fmt & chars_format::fixed)) {
+        // The exponential part is invalid for scientific notation, so it must
+        // be a trailing token for fixed notation. However, fixed notation is
+        // disabled, so report a scientific notation error.
+        return report_parse_error<UC>(p, parse_error::missing_exponential_part);
+      }
+      // Otherwise, we will be ignoring the 'e'.
+      p = location_of_e;
+    } else {
+      while ((p != pend) && is_integer(*p)) {
+        uint8_t digit = uint8_t(*p - UC('0'));
+        if (exp_number < 0x10000000) {
+          exp_number = 10 * exp_number + digit;
+        }
+        ++p;
+      }
+      if (neg_exp) {
+        exp_number = -exp_number;
+      }
+      exponent += exp_number;
+    }
+  } else {
+    // If it scientific and not fixed, we have to bail out.
+    if (uint64_t(fmt & chars_format::scientific) &&
+        !uint64_t(fmt & chars_format::fixed)) {
+      return report_parse_error<UC>(p, parse_error::missing_exponential_part);
+    }
+  }
+  answer.lastmatch = p;
+  answer.valid = true;
+
+  // If we frequently had to deal with long strings of digits,
+  // we could extend our code by using a 128-bit integer instead
+  // of a 64-bit integer. However, this is uncommon.
+  //
+  // We can deal with up to 19 digits.
+  if (digit_count > 19) { // this is uncommon
+    // It is possible that the integer had an overflow.
+    // We have to handle the case where we have 0.0000somenumber.
+    // We need to be mindful of the case where we only have zeroes...
+    // E.g., 0.000000000...000.
+    UC const *start = start_digits;
+    while ((start != pend) && (*start == UC('0') || *start == decimal_point)) {
+      if (*start == UC('0')) {
+        digit_count--;
+      }
+      start++;
+    }
+
+    if (digit_count > 19) {
+      answer.too_many_digits = true;
+      // The truncation recompute below reads the integer/fraction spans. When
+      // store_spans is false we didn't materialize them, so just flag
+      // too_many_digits; the caller re-parses with store_spans=true to obtain
+      // the corrected mantissa/exponent before taking the slow path.
+      if (store_spans) {
+        // Let us start again, this time, avoiding overflows.
+        // We don't need to call if is_integer, since we use the
+        // pre-tokenized spans from above.
+        i = 0;
+        p = answer.integer.ptr;
+        UC const *int_end = p + answer.integer.len();
+        uint64_t const minimal_nineteen_digit_integer{1000000000000000000};
+        while ((i < minimal_nineteen_digit_integer) && (p != int_end)) {
+          i = i * 10 + uint64_t(*p - UC('0'));
+          ++p;
+        }
+        if (i >= minimal_nineteen_digit_integer) { // We have a big integer
+          exponent = end_of_integer_part - p + exp_number;
+        } else { // We have a value with a fractional component.
+          p = answer.fraction.ptr;
+          UC const *frac_end = p + answer.fraction.len();
+          while ((i < minimal_nineteen_digit_integer) && (p != frac_end)) {
+            i = i * 10 + uint64_t(*p - UC('0'));
+            ++p;
+          }
+          exponent = answer.fraction.ptr - p + exp_number;
+        }
+        // We have now corrected both exponent and i, to a truncated value
+      }
+    }
+  }
+  answer.exponent = exponent;
+  answer.mantissa = i;
+  return answer;
+}
+
+template <typename T, typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+parse_int_string(UC const *p, UC const *pend, T &value,
+                 parse_options_t<UC> options) {
+  chars_format const fmt = detail::adjust_for_feature_macros(options.format);
+  int const base = options.base;
+
+  from_chars_result_t<UC> answer;
+
+  UC const *const first = p;
+
+  bool const negative = (*p == UC('-'));
+#ifdef SIMDJSON_FASTFLOAT_VISUAL_STUDIO
+#pragma warning(push)
+#pragma warning(disable : 4127)
+#endif
+  if (!std::is_signed<T>::value && negative) {
+#ifdef SIMDJSON_FASTFLOAT_VISUAL_STUDIO
+#pragma warning(pop)
+#endif
+    answer.ec = std::errc::invalid_argument;
+    answer.ptr = first;
+    return answer;
+  }
+  if ((*p == UC('-')) ||
+      (uint64_t(fmt & chars_format::allow_leading_plus) && (*p == UC('+')))) {
+    ++p;
+  }
+
+  UC const *const start_num = p;
+
+  while (p != pend && *p == UC('0')) {
+    ++p;
+  }
+
+  bool const has_leading_zeros = p > start_num;
+
+  UC const *const start_digits = p;
+
+  SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(
+      (std::is_same<T, std::uint8_t>::value && sizeof(UC) == 1)) {
+    if (base == 10) {
+      const size_t len = static_cast<size_t>(pend - p);
+      if (len == 0) {
+        if (has_leading_zeros) {
+          value = 0;
+          answer.ec = std::errc();
+          answer.ptr = p;
+        } else {
+          answer.ec = std::errc::invalid_argument;
+          answer.ptr = first;
+        }
+        return answer;
+      }
+
+      uint32_t digits;
+
+#if SIMDJSON_FASTFLOAT_HAS_IS_CONSTANT_EVALUATED && SIMDJSON_FASTFLOAT_HAS_BIT_CAST
+      if (std::is_constant_evaluated()) {
+        uint8_t str[4]{};
+        for (size_t j = 0; j < 4 && j < len; ++j) {
+          str[j] = static_cast<uint8_t>(p[j]);
+        }
+        digits = std::bit_cast<uint32_t>(str);
+#if SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN
+        digits = byteswap_32(digits);
+#endif
+      }
+#else
+      if (false) {
+      }
+#endif
+      else if (len >= 4) {
+        ::memcpy(&digits, p, 4);
+#if SIMDJSON_FASTFLOAT_IS_BIG_ENDIAN
+        digits = byteswap_32(digits);
+#endif
+      } else {
+        uint32_t b0 = static_cast<uint8_t>(p[0]);
+        uint32_t b1 = (len > 1) ? static_cast<uint8_t>(p[1]) : 0xFFu;
+        uint32_t b2 = (len > 2) ? static_cast<uint8_t>(p[2]) : 0xFFu;
+        uint32_t b3 = 0xFFu;
+        digits = b0 | (b1 << 8) | (b2 << 16) | (b3 << 24);
+      }
+
+      uint32_t magic =
+          ((digits + 0x46464646u) | (digits - 0x30303030u)) & 0x80808080u;
+      uint32_t tz =
+          static_cast<uint32_t>(countr_zero_32(magic)); // 7, 15, 23, 31, or 32
+      uint32_t nd = (tz == 32) ? 4 : (tz >> 3);
+      nd = static_cast<uint32_t>(nd < len ? nd : len);
+      if (nd == 0) {
+        if (has_leading_zeros) {
+          value = 0;
+          answer.ec = std::errc();
+          answer.ptr = p;
+          return answer;
+        }
+        answer.ec = std::errc::invalid_argument;
+        answer.ptr = first;
+        return answer;
+      }
+      if (nd > 3) {
+        const UC *q = p + nd;
+        size_t rem = len - nd;
+        while (rem) {
+          if (*q < UC('0') || *q > UC('9'))
+            break;
+          ++q;
+          --rem;
+        }
+        answer.ec = std::errc::result_out_of_range;
+        answer.ptr = q;
+        return answer;
+      }
+
+      digits ^= 0x30303030u;
+      digits <<= ((4 - nd) * 8);
+
+      uint32_t check = ((digits >> 24) & 0xff) | ((digits >> 8) & 0xff00) |
+                       ((digits << 8) & 0xff0000);
+      if (check > 0x00020505) {
+        answer.ec = std::errc::result_out_of_range;
+        answer.ptr = p + nd;
+        return answer;
+      }
+      value = static_cast<uint8_t>((0x640a01 * digits) >> 24);
+      answer.ec = std::errc();
+      answer.ptr = p + nd;
+      return answer;
+    }
+  }
+
+  SIMDJSON_FASTFLOAT_IF_CONSTEXPR17(
+      (std::is_same<T, std::uint16_t>::value && sizeof(UC) == 1)) {
+    if (base == 10) {
+      const size_t len = size_t(pend - p);
+      if (len == 0) {
+        if (has_leading_zeros) {
+          value = 0;
+          answer.ec = std::errc();
+          answer.ptr = p;
+        } else {
+          answer.ec = std::errc::invalid_argument;
+          answer.ptr = first;
+        }
+        return answer;
+      }
+
+      if (len >= 4) {
+        uint32_t digits = read4_to_u32(p);
+        if (is_made_of_four_digits_fast(digits)) {
+          uint32_t v = parse_four_digits_unrolled(digits);
+          if (len >= 5 && is_integer(p[4])) {
+            v = v * 10 + uint32_t(p[4] - '0');
+            if (len >= 6 && is_integer(p[5])) {
+              answer.ec = std::errc::result_out_of_range;
+              const UC *q = p + 5;
+              while (q != pend && is_integer(*q)) {
+                q++;
+              }
+              answer.ptr = q;
+              return answer;
+            }
+            if (v > 65535) {
+              answer.ec = std::errc::result_out_of_range;
+              answer.ptr = p + 5;
+              return answer;
+            }
+            value = uint16_t(v);
+            answer.ec = std::errc();
+            answer.ptr = p + 5;
+            return answer;
+          }
+          // 4 digits
+          value = uint16_t(v);
+          answer.ec = std::errc();
+          answer.ptr = p + 4;
+          return answer;
+        }
+      }
+    }
+  }
+
+  uint64_t i = 0;
+  if (base == 10) {
+    loop_parse_if_eight_digits(p, pend, i); // use SIMD if possible
+  }
+  while (p != pend) {
+    uint8_t digit = ch_to_digit(*p);
+    if (digit >= base) {
+      break;
+    }
+    i = uint64_t(base) * i + digit; // might overflow, check this later
+    p++;
+  }
+
+  size_t digit_count = size_t(p - start_digits);
+
+  if (digit_count == 0) {
+    if (has_leading_zeros) {
+      value = 0;
+      answer.ec = std::errc();
+      answer.ptr = p;
+    } else {
+      answer.ec = std::errc::invalid_argument;
+      answer.ptr = first;
+    }
+    return answer;
+  }
+
+  answer.ptr = p;
+
+  // check u64 overflow
+  size_t max_digits = max_digits_u64(base);
+  if (digit_count > max_digits) {
+    answer.ec = std::errc::result_out_of_range;
+    return answer;
+  }
+  // this check can be eliminated for all other types, but they will all require
+  // a max_digits(base) equivalent
+  if (digit_count == max_digits) {
+    // At the max_digits boundary the accumulator `i` may have wrapped around
+    // 2^64. A plain `i < min_safe_u64(base)` test is not sufficient: for any
+    // base whose max_digits-length range exceeds 2^64 (base 10 reaches
+    // ~5.4 * 2^64 at 20 digits) the value can wrap a whole multiple of 2^64 and
+    // land back above min_safe, slipping through. Decide exactly in O(1) using
+    // the leading digit, following the approach used in simdjson:
+    //   ms   == min_safe_u64(base) == base^(max_digits-1), the smallest
+    //           max_digits-length value.
+    //   dmax == the largest leading digit whose number can still fit in u64.
+    // The leading-digit band [d*ms, (d+1)*ms) has width ms < 2^64, so within
+    // the single band where d == dmax the value straddles 2^64 at most once,
+    // and a single threshold separates wrapped from non-wrapped values. A
+    // leading digit above dmax always overflows; below dmax always fits.
+    uint64_t const ms = min_safe_u64(base);
+    uint64_t const dmax = (std::numeric_limits<uint64_t>::max)() / ms;
+    uint64_t const lead = ch_to_digit(*start_digits);
+    if (lead > dmax || (lead == dmax && i < dmax * ms)) {
+      answer.ec = std::errc::result_out_of_range;
+      return answer;
+    }
+  }
+
+  // check other types overflow
+  if (!std::is_same<T, uint64_t>::value) {
+    if (i > uint64_t((std::numeric_limits<T>::max)()) + uint64_t(negative)) {
+      answer.ec = std::errc::result_out_of_range;
+      return answer;
+    }
+  }
+
+  if (negative) {
+#ifdef SIMDJSON_FASTFLOAT_VISUAL_STUDIO
+#pragma warning(push)
+#pragma warning(disable : 4146)
+#endif
+    // this weird workaround is required because:
+    // - converting unsigned to signed when its value is greater than signed max
+    // is UB pre-C++23.
+    // - reinterpret_casting (~i + 1) would work, but it is not constexpr
+    // this is always optimized into a neg instruction (note: T is an integer
+    // type)
+    value = T(-(std::numeric_limits<T>::max)() -
+              T(i - uint64_t((std::numeric_limits<T>::max)())));
+#ifdef SIMDJSON_FASTFLOAT_VISUAL_STUDIO
+#pragma warning(pop)
+#endif
+  } else {
+    value = T(i);
+  }
+
+  answer.ec = std::errc();
+  return answer;
+}
+
+} // namespace simdjson_fast_float
+
+#endif
+
+#ifndef SIMDJSON_FASTFLOAT_FAST_TABLE_H
+#define SIMDJSON_FASTFLOAT_FAST_TABLE_H
+
+#include <cstdint>
+
+namespace simdjson_fast_float {
+
+/**
+ * When mapping numbers from decimal to binary,
+ * we go from w * 10^q to m * 2^p but we have
+ * 10^q = 5^q * 2^q, so effectively
+ * we are trying to match
+ * w * 2^q * 5^q to m * 2^p. Thus the powers of two
+ * are not a concern since they can be represented
+ * exactly using the binary notation, only the powers of five
+ * affect the binary significand.
+ */
+
+/**
+ * The smallest non-zero float (binary64) is 2^-1074.
+ * We take as input numbers of the form w x 10^q where w < 2^64.
+ * We have that w * 10^-343  <  2^(64-344) 5^-343 < 2^-1076.
+ * However, we have that
+ * (2^64-1) * 10^-342 =  (2^64-1) * 2^-342 * 5^-342 > 2^-1074.
+ * Thus it is possible for a number of the form w * 10^-342 where
+ * w is a 64-bit value to be a non-zero floating-point number.
+ *********
+ * Any number of form w * 10^309 where w>= 1 is going to be
+ * infinite in binary64 so we never need to worry about powers
+ * of 5 greater than 308.
+ */
+template <class unused = void> struct powers_template {
+
+  constexpr static int smallest_power_of_five =
+      binary_format<double>::smallest_power_of_ten();
+  constexpr static int largest_power_of_five =
+      binary_format<double>::largest_power_of_ten();
+  constexpr static int number_of_entries =
+      2 * (largest_power_of_five - smallest_power_of_five + 1);
+  // Powers of five from 5^-342 all the way to 5^308 rounded toward one.
+  constexpr static uint64_t power_of_five_128[number_of_entries] = {
+      0xeef453d6923bd65a, 0x113faa2906a13b3f,
+      0x9558b4661b6565f8, 0x4ac7ca59a424c507,
+      0xbaaee17fa23ebf76, 0x5d79bcf00d2df649,
+      0xe95a99df8ace6f53, 0xf4d82c2c107973dc,
+      0x91d8a02bb6c10594, 0x79071b9b8a4be869,
+      0xb64ec836a47146f9, 0x9748e2826cdee284,
+      0xe3e27a444d8d98b7, 0xfd1b1b2308169b25,
+      0x8e6d8c6ab0787f72, 0xfe30f0f5e50e20f7,
+      0xb208ef855c969f4f, 0xbdbd2d335e51a935,
+      0xde8b2b66b3bc4723, 0xad2c788035e61382,
+      0x8b16fb203055ac76, 0x4c3bcb5021afcc31,
+      0xaddcb9e83c6b1793, 0xdf4abe242a1bbf3d,
+      0xd953e8624b85dd78, 0xd71d6dad34a2af0d,
+      0x87d4713d6f33aa6b, 0x8672648c40e5ad68,
+      0xa9c98d8ccb009506, 0x680efdaf511f18c2,
+      0xd43bf0effdc0ba48, 0x212bd1b2566def2,
+      0x84a57695fe98746d, 0x14bb630f7604b57,
+      0xa5ced43b7e3e9188, 0x419ea3bd35385e2d,
+      0xcf42894a5dce35ea, 0x52064cac828675b9,
+      0x818995ce7aa0e1b2, 0x7343efebd1940993,
+      0xa1ebfb4219491a1f, 0x1014ebe6c5f90bf8,
+      0xca66fa129f9b60a6, 0xd41a26e077774ef6,
+      0xfd00b897478238d0, 0x8920b098955522b4,
+      0x9e20735e8cb16382, 0x55b46e5f5d5535b0,
+      0xc5a890362fddbc62, 0xeb2189f734aa831d,
+      0xf712b443bbd52b7b, 0xa5e9ec7501d523e4,
+      0x9a6bb0aa55653b2d, 0x47b233c92125366e,
+      0xc1069cd4eabe89f8, 0x999ec0bb696e840a,
+      0xf148440a256e2c76, 0xc00670ea43ca250d,
+      0x96cd2a865764dbca, 0x380406926a5e5728,
+      0xbc807527ed3e12bc, 0xc605083704f5ecf2,
+      0xeba09271e88d976b, 0xf7864a44c633682e,
+      0x93445b8731587ea3, 0x7ab3ee6afbe0211d,
+      0xb8157268fdae9e4c, 0x5960ea05bad82964,
+      0xe61acf033d1a45df, 0x6fb92487298e33bd,
+      0x8fd0c16206306bab, 0xa5d3b6d479f8e056,
+      0xb3c4f1ba87bc8696, 0x8f48a4899877186c,
+      0xe0b62e2929aba83c, 0x331acdabfe94de87,
+      0x8c71dcd9ba0b4925, 0x9ff0c08b7f1d0b14,
+      0xaf8e5410288e1b6f, 0x7ecf0ae5ee44dd9,
+      0xdb71e91432b1a24a, 0xc9e82cd9f69d6150,
+      0x892731ac9faf056e, 0xbe311c083a225cd2,
+      0xab70fe17c79ac6ca, 0x6dbd630a48aaf406,
+      0xd64d3d9db981787d, 0x92cbbccdad5b108,
+      0x85f0468293f0eb4e, 0x25bbf56008c58ea5,
+      0xa76c582338ed2621, 0xaf2af2b80af6f24e,
+      0xd1476e2c07286faa, 0x1af5af660db4aee1,
+      0x82cca4db847945ca, 0x50d98d9fc890ed4d,
+      0xa37fce126597973c, 0xe50ff107bab528a0,
+      0xcc5fc196fefd7d0c, 0x1e53ed49a96272c8,
+      0xff77b1fcbebcdc4f, 0x25e8e89c13bb0f7a,
+      0x9faacf3df73609b1, 0x77b191618c54e9ac,
+      0xc795830d75038c1d, 0xd59df5b9ef6a2417,
+      0xf97ae3d0d2446f25, 0x4b0573286b44ad1d,
+      0x9becce62836ac577, 0x4ee367f9430aec32,
+      0xc2e801fb244576d5, 0x229c41f793cda73f,
+      0xf3a20279ed56d48a, 0x6b43527578c1110f,
+      0x9845418c345644d6, 0x830a13896b78aaa9,
+      0xbe5691ef416bd60c, 0x23cc986bc656d553,
+      0xedec366b11c6cb8f, 0x2cbfbe86b7ec8aa8,
+      0x94b3a202eb1c3f39, 0x7bf7d71432f3d6a9,
+      0xb9e08a83a5e34f07, 0xdaf5ccd93fb0cc53,
+      0xe858ad248f5c22c9, 0xd1b3400f8f9cff68,
+      0x91376c36d99995be, 0x23100809b9c21fa1,
+      0xb58547448ffffb2d, 0xabd40a0c2832a78a,
+      0xe2e69915b3fff9f9, 0x16c90c8f323f516c,
+      0x8dd01fad907ffc3b, 0xae3da7d97f6792e3,
+      0xb1442798f49ffb4a, 0x99cd11cfdf41779c,
+      0xdd95317f31c7fa1d, 0x40405643d711d583,
+      0x8a7d3eef7f1cfc52, 0x482835ea666b2572,
+      0xad1c8eab5ee43b66, 0xda3243650005eecf,
+      0xd863b256369d4a40, 0x90bed43e40076a82,
+      0x873e4f75e2224e68, 0x5a7744a6e804a291,
+      0xa90de3535aaae202, 0x711515d0a205cb36,
+      0xd3515c2831559a83, 0xd5a5b44ca873e03,
+      0x8412d9991ed58091, 0xe858790afe9486c2,
+      0xa5178fff668ae0b6, 0x626e974dbe39a872,
+      0xce5d73ff402d98e3, 0xfb0a3d212dc8128f,
+      0x80fa687f881c7f8e, 0x7ce66634bc9d0b99,
+      0xa139029f6a239f72, 0x1c1fffc1ebc44e80,
+      0xc987434744ac874e, 0xa327ffb266b56220,
+      0xfbe9141915d7a922, 0x4bf1ff9f0062baa8,
+      0x9d71ac8fada6c9b5, 0x6f773fc3603db4a9,
+      0xc4ce17b399107c22, 0xcb550fb4384d21d3,
+      0xf6019da07f549b2b, 0x7e2a53a146606a48,
+      0x99c102844f94e0fb, 0x2eda7444cbfc426d,
+      0xc0314325637a1939, 0xfa911155fefb5308,
+      0xf03d93eebc589f88, 0x793555ab7eba27ca,
+      0x96267c7535b763b5, 0x4bc1558b2f3458de,
+      0xbbb01b9283253ca2, 0x9eb1aaedfb016f16,
+      0xea9c227723ee8bcb, 0x465e15a979c1cadc,
+      0x92a1958a7675175f, 0xbfacd89ec191ec9,
+      0xb749faed14125d36, 0xcef980ec671f667b,
+      0xe51c79a85916f484, 0x82b7e12780e7401a,
+      0x8f31cc0937ae58d2, 0xd1b2ecb8b0908810,
+      0xb2fe3f0b8599ef07, 0x861fa7e6dcb4aa15,
+      0xdfbdcece67006ac9, 0x67a791e093e1d49a,
+      0x8bd6a141006042bd, 0xe0c8bb2c5c6d24e0,
+      0xaecc49914078536d, 0x58fae9f773886e18,
+      0xda7f5bf590966848, 0xaf39a475506a899e,
+      0x888f99797a5e012d, 0x6d8406c952429603,
+      0xaab37fd7d8f58178, 0xc8e5087ba6d33b83,
+      0xd5605fcdcf32e1d6, 0xfb1e4a9a90880a64,
+      0x855c3be0a17fcd26, 0x5cf2eea09a55067f,
+      0xa6b34ad8c9dfc06f, 0xf42faa48c0ea481e,
+      0xd0601d8efc57b08b, 0xf13b94daf124da26,
+      0x823c12795db6ce57, 0x76c53d08d6b70858,
+      0xa2cb1717b52481ed, 0x54768c4b0c64ca6e,
+      0xcb7ddcdda26da268, 0xa9942f5dcf7dfd09,
+      0xfe5d54150b090b02, 0xd3f93b35435d7c4c,
+      0x9efa548d26e5a6e1, 0xc47bc5014a1a6daf,
+      0xc6b8e9b0709f109a, 0x359ab6419ca1091b,
+      0xf867241c8cc6d4c0, 0xc30163d203c94b62,
+      0x9b407691d7fc44f8, 0x79e0de63425dcf1d,
+      0xc21094364dfb5636, 0x985915fc12f542e4,
+      0xf294b943e17a2bc4, 0x3e6f5b7b17b2939d,
+      0x979cf3ca6cec5b5a, 0xa705992ceecf9c42,
+      0xbd8430bd08277231, 0x50c6ff782a838353,
+      0xece53cec4a314ebd, 0xa4f8bf5635246428,
+      0x940f4613ae5ed136, 0x871b7795e136be99,
+      0xb913179899f68584, 0x28e2557b59846e3f,
+      0xe757dd7ec07426e5, 0x331aeada2fe589cf,
+      0x9096ea6f3848984f, 0x3ff0d2c85def7621,
+      0xb4bca50b065abe63, 0xfed077a756b53a9,
+      0xe1ebce4dc7f16dfb, 0xd3e8495912c62894,
+      0x8d3360f09cf6e4bd, 0x64712dd7abbbd95c,
+      0xb080392cc4349dec, 0xbd8d794d96aacfb3,
+      0xdca04777f541c567, 0xecf0d7a0fc5583a0,
+      0x89e42caaf9491b60, 0xf41686c49db57244,
+      0xac5d37d5b79b6239, 0x311c2875c522ced5,
+      0xd77485cb25823ac7, 0x7d633293366b828b,
+      0x86a8d39ef77164bc, 0xae5dff9c02033197,
+      0xa8530886b54dbdeb, 0xd9f57f830283fdfc,
+      0xd267caa862a12d66, 0xd072df63c324fd7b,
+      0x8380dea93da4bc60, 0x4247cb9e59f71e6d,
+      0xa46116538d0deb78, 0x52d9be85f074e608,
+      0xcd795be870516656, 0x67902e276c921f8b,
+      0x806bd9714632dff6, 0xba1cd8a3db53b6,
+      0xa086cfcd97bf97f3, 0x80e8a40eccd228a4,
+      0xc8a883c0fdaf7df0, 0x6122cd128006b2cd,
+      0xfad2a4b13d1b5d6c, 0x796b805720085f81,
+      0x9cc3a6eec6311a63, 0xcbe3303674053bb0,
+      0xc3f490aa77bd60fc, 0xbedbfc4411068a9c,
+      0xf4f1b4d515acb93b, 0xee92fb5515482d44,
+      0x991711052d8bf3c5, 0x751bdd152d4d1c4a,
+      0xbf5cd54678eef0b6, 0xd262d45a78a0635d,
+      0xef340a98172aace4, 0x86fb897116c87c34,
+      0x9580869f0e7aac0e, 0xd45d35e6ae3d4da0,
+      0xbae0a846d2195712, 0x8974836059cca109,
+      0xe998d258869facd7, 0x2bd1a438703fc94b,
+      0x91ff83775423cc06, 0x7b6306a34627ddcf,
+      0xb67f6455292cbf08, 0x1a3bc84c17b1d542,
+      0xe41f3d6a7377eeca, 0x20caba5f1d9e4a93,
+      0x8e938662882af53e, 0x547eb47b7282ee9c,
+      0xb23867fb2a35b28d, 0xe99e619a4f23aa43,
+      0xdec681f9f4c31f31, 0x6405fa00e2ec94d4,
+      0x8b3c113c38f9f37e, 0xde83bc408dd3dd04,
+      0xae0b158b4738705e, 0x9624ab50b148d445,
+      0xd98ddaee19068c76, 0x3badd624dd9b0957,
+      0x87f8a8d4cfa417c9, 0xe54ca5d70a80e5d6,
+      0xa9f6d30a038d1dbc, 0x5e9fcf4ccd211f4c,
+      0xd47487cc8470652b, 0x7647c3200069671f,
+      0x84c8d4dfd2c63f3b, 0x29ecd9f40041e073,
+      0xa5fb0a17c777cf09, 0xf468107100525890,
+      0xcf79cc9db955c2cc, 0x7182148d4066eeb4,
+      0x81ac1fe293d599bf, 0xc6f14cd848405530,
+      0xa21727db38cb002f, 0xb8ada00e5a506a7c,
+      0xca9cf1d206fdc03b, 0xa6d90811f0e4851c,
+      0xfd442e4688bd304a, 0x908f4a166d1da663,
+      0x9e4a9cec15763e2e, 0x9a598e4e043287fe,
+      0xc5dd44271ad3cdba, 0x40eff1e1853f29fd,
+      0xf7549530e188c128, 0xd12bee59e68ef47c,
+      0x9a94dd3e8cf578b9, 0x82bb74f8301958ce,
+      0xc13a148e3032d6e7, 0xe36a52363c1faf01,
+      0xf18899b1bc3f8ca1, 0xdc44e6c3cb279ac1,
+      0x96f5600f15a7b7e5, 0x29ab103a5ef8c0b9,
+      0xbcb2b812db11a5de, 0x7415d448f6b6f0e7,
+      0xebdf661791d60f56, 0x111b495b3464ad21,
+      0x936b9fcebb25c995, 0xcab10dd900beec34,
+      0xb84687c269ef3bfb, 0x3d5d514f40eea742,
+      0xe65829b3046b0afa, 0xcb4a5a3112a5112,
+      0x8ff71a0fe2c2e6dc, 0x47f0e785eaba72ab,
+      0xb3f4e093db73a093, 0x59ed216765690f56,
+      0xe0f218b8d25088b8, 0x306869c13ec3532c,
+      0x8c974f7383725573, 0x1e414218c73a13fb,
+      0xafbd2350644eeacf, 0xe5d1929ef90898fa,
+      0xdbac6c247d62a583, 0xdf45f746b74abf39,
+      0x894bc396ce5da772, 0x6b8bba8c328eb783,
+      0xab9eb47c81f5114f, 0x66ea92f3f326564,
+      0xd686619ba27255a2, 0xc80a537b0efefebd,
+      0x8613fd0145877585, 0xbd06742ce95f5f36,
+      0xa798fc4196e952e7, 0x2c48113823b73704,
+      0xd17f3b51fca3a7a0, 0xf75a15862ca504c5,
+      0x82ef85133de648c4, 0x9a984d73dbe722fb,
+      0xa3ab66580d5fdaf5, 0xc13e60d0d2e0ebba,
+      0xcc963fee10b7d1b3, 0x318df905079926a8,
+      0xffbbcfe994e5c61f, 0xfdf17746497f7052,
+      0x9fd561f1fd0f9bd3, 0xfeb6ea8bedefa633,
+      0xc7caba6e7c5382c8, 0xfe64a52ee96b8fc0,
+      0xf9bd690a1b68637b, 0x3dfdce7aa3c673b0,
+      0x9c1661a651213e2d, 0x6bea10ca65c084e,
+      0xc31bfa0fe5698db8, 0x486e494fcff30a62,
+      0xf3e2f893dec3f126, 0x5a89dba3c3efccfa,
+      0x986ddb5c6b3a76b7, 0xf89629465a75e01c,
+      0xbe89523386091465, 0xf6bbb397f1135823,
+      0xee2ba6c0678b597f, 0x746aa07ded582e2c,
+      0x94db483840b717ef, 0xa8c2a44eb4571cdc,
+      0xba121a4650e4ddeb, 0x92f34d62616ce413,
+      0xe896a0d7e51e1566, 0x77b020baf9c81d17,
+      0x915e2486ef32cd60, 0xace1474dc1d122e,
+      0xb5b5ada8aaff80b8, 0xd819992132456ba,
+      0xe3231912d5bf60e6, 0x10e1fff697ed6c69,
+      0x8df5efabc5979c8f, 0xca8d3ffa1ef463c1,
+      0xb1736b96b6fd83b3, 0xbd308ff8a6b17cb2,
+      0xddd0467c64bce4a0, 0xac7cb3f6d05ddbde,
+      0x8aa22c0dbef60ee4, 0x6bcdf07a423aa96b,
+      0xad4ab7112eb3929d, 0x86c16c98d2c953c6,
+      0xd89d64d57a607744, 0xe871c7bf077ba8b7,
+      0x87625f056c7c4a8b, 0x11471cd764ad4972,
+      0xa93af6c6c79b5d2d, 0xd598e40d3dd89bcf,
+      0xd389b47879823479, 0x4aff1d108d4ec2c3,
+      0x843610cb4bf160cb, 0xcedf722a585139ba,
+      0xa54394fe1eedb8fe, 0xc2974eb4ee658828,
+      0xce947a3da6a9273e, 0x733d226229feea32,
+      0x811ccc668829b887, 0x806357d5a3f525f,
+      0xa163ff802a3426a8, 0xca07c2dcb0cf26f7,
+      0xc9bcff6034c13052, 0xfc89b393dd02f0b5,
+      0xfc2c3f3841f17c67, 0xbbac2078d443ace2,
+      0x9d9ba7832936edc0, 0xd54b944b84aa4c0d,
+      0xc5029163f384a931, 0xa9e795e65d4df11,
+      0xf64335bcf065d37d, 0x4d4617b5ff4a16d5,
+      0x99ea0196163fa42e, 0x504bced1bf8e4e45,
+      0xc06481fb9bcf8d39, 0xe45ec2862f71e1d6,
+      0xf07da27a82c37088, 0x5d767327bb4e5a4c,
+      0x964e858c91ba2655, 0x3a6a07f8d510f86f,
+      0xbbe226efb628afea, 0x890489f70a55368b,
+      0xeadab0aba3b2dbe5, 0x2b45ac74ccea842e,
+      0x92c8ae6b464fc96f, 0x3b0b8bc90012929d,
+      0xb77ada0617e3bbcb, 0x9ce6ebb40173744,
+      0xe55990879ddcaabd, 0xcc420a6a101d0515,
+      0x8f57fa54c2a9eab6, 0x9fa946824a12232d,
+      0xb32df8e9f3546564, 0x47939822dc96abf9,
+      0xdff9772470297ebd, 0x59787e2b93bc56f7,
+      0x8bfbea76c619ef36, 0x57eb4edb3c55b65a,
+      0xaefae51477a06b03, 0xede622920b6b23f1,
+      0xdab99e59958885c4, 0xe95fab368e45eced,
+      0x88b402f7fd75539b, 0x11dbcb0218ebb414,
+      0xaae103b5fcd2a881, 0xd652bdc29f26a119,
+      0xd59944a37c0752a2, 0x4be76d3346f0495f,
+      0x857fcae62d8493a5, 0x6f70a4400c562ddb,
+      0xa6dfbd9fb8e5b88e, 0xcb4ccd500f6bb952,
+      0xd097ad07a71f26b2, 0x7e2000a41346a7a7,
+      0x825ecc24c873782f, 0x8ed400668c0c28c8,
+      0xa2f67f2dfa90563b, 0x728900802f0f32fa,
+      0xcbb41ef979346bca, 0x4f2b40a03ad2ffb9,
+      0xfea126b7d78186bc, 0xe2f610c84987bfa8,
+      0x9f24b832e6b0f436, 0xdd9ca7d2df4d7c9,
+      0xc6ede63fa05d3143, 0x91503d1c79720dbb,
+      0xf8a95fcf88747d94, 0x75a44c6397ce912a,
+      0x9b69dbe1b548ce7c, 0xc986afbe3ee11aba,
+      0xc24452da229b021b, 0xfbe85badce996168,
+      0xf2d56790ab41c2a2, 0xfae27299423fb9c3,
+      0x97c560ba6b0919a5, 0xdccd879fc967d41a,
+      0xbdb6b8e905cb600f, 0x5400e987bbc1c920,
+      0xed246723473e3813, 0x290123e9aab23b68,
+      0x9436c0760c86e30b, 0xf9a0b6720aaf6521,
+      0xb94470938fa89bce, 0xf808e40e8d5b3e69,
+      0xe7958cb87392c2c2, 0xb60b1d1230b20e04,
+      0x90bd77f3483bb9b9, 0xb1c6f22b5e6f48c2,
+      0xb4ecd5f01a4aa828, 0x1e38aeb6360b1af3,
+      0xe2280b6c20dd5232, 0x25c6da63c38de1b0,
+      0x8d590723948a535f, 0x579c487e5a38ad0e,
+      0xb0af48ec79ace837, 0x2d835a9df0c6d851,
+      0xdcdb1b2798182244, 0xf8e431456cf88e65,
+      0x8a08f0f8bf0f156b, 0x1b8e9ecb641b58ff,
+      0xac8b2d36eed2dac5, 0xe272467e3d222f3f,
+      0xd7adf884aa879177, 0x5b0ed81dcc6abb0f,
+      0x86ccbb52ea94baea, 0x98e947129fc2b4e9,
+      0xa87fea27a539e9a5, 0x3f2398d747b36224,
+      0xd29fe4b18e88640e, 0x8eec7f0d19a03aad,
+      0x83a3eeeef9153e89, 0x1953cf68300424ac,
+      0xa48ceaaab75a8e2b, 0x5fa8c3423c052dd7,
+      0xcdb02555653131b6, 0x3792f412cb06794d,
+      0x808e17555f3ebf11, 0xe2bbd88bbee40bd0,
+      0xa0b19d2ab70e6ed6, 0x5b6aceaeae9d0ec4,
+      0xc8de047564d20a8b, 0xf245825a5a445275,
+      0xfb158592be068d2e, 0xeed6e2f0f0d56712,
+      0x9ced737bb6c4183d, 0x55464dd69685606b,
+      0xc428d05aa4751e4c, 0xaa97e14c3c26b886,
+      0xf53304714d9265df, 0xd53dd99f4b3066a8,
+      0x993fe2c6d07b7fab, 0xe546a8038efe4029,
+      0xbf8fdb78849a5f96, 0xde98520472bdd033,
+      0xef73d256a5c0f77c, 0x963e66858f6d4440,
+      0x95a8637627989aad, 0xdde7001379a44aa8,
+      0xbb127c53b17ec159, 0x5560c018580d5d52,
+      0xe9d71b689dde71af, 0xaab8f01e6e10b4a6,
+      0x9226712162ab070d, 0xcab3961304ca70e8,
+      0xb6b00d69bb55c8d1, 0x3d607b97c5fd0d22,
+      0xe45c10c42a2b3b05, 0x8cb89a7db77c506a,
+      0x8eb98a7a9a5b04e3, 0x77f3608e92adb242,
+      0xb267ed1940f1c61c, 0x55f038b237591ed3,
+      0xdf01e85f912e37a3, 0x6b6c46dec52f6688,
+      0x8b61313bbabce2c6, 0x2323ac4b3b3da015,
+      0xae397d8aa96c1b77, 0xabec975e0a0d081a,
+      0xd9c7dced53c72255, 0x96e7bd358c904a21,
+      0x881cea14545c7575, 0x7e50d64177da2e54,
+      0xaa242499697392d2, 0xdde50bd1d5d0b9e9,
+      0xd4ad2dbfc3d07787, 0x955e4ec64b44e864,
+      0x84ec3c97da624ab4, 0xbd5af13bef0b113e,
+      0xa6274bbdd0fadd61, 0xecb1ad8aeacdd58e,
+      0xcfb11ead453994ba, 0x67de18eda5814af2,
+      0x81ceb32c4b43fcf4, 0x80eacf948770ced7,
+      0xa2425ff75e14fc31, 0xa1258379a94d028d,
+      0xcad2f7f5359a3b3e, 0x96ee45813a04330,
+      0xfd87b5f28300ca0d, 0x8bca9d6e188853fc,
+      0x9e74d1b791e07e48, 0x775ea264cf55347e,
+      0xc612062576589dda, 0x95364afe032a819e,
+      0xf79687aed3eec551, 0x3a83ddbd83f52205,
+      0x9abe14cd44753b52, 0xc4926a9672793543,
+      0xc16d9a0095928a27, 0x75b7053c0f178294,
+      0xf1c90080baf72cb1, 0x5324c68b12dd6339,
+      0x971da05074da7bee, 0xd3f6fc16ebca5e04,
+      0xbce5086492111aea, 0x88f4bb1ca6bcf585,
+      0xec1e4a7db69561a5, 0x2b31e9e3d06c32e6,
+      0x9392ee8e921d5d07, 0x3aff322e62439fd0,
+      0xb877aa3236a4b449, 0x9befeb9fad487c3,
+      0xe69594bec44de15b, 0x4c2ebe687989a9b4,
+      0x901d7cf73ab0acd9, 0xf9d37014bf60a11,
+      0xb424dc35095cd80f, 0x538484c19ef38c95,
+      0xe12e13424bb40e13, 0x2865a5f206b06fba,
+      0x8cbccc096f5088cb, 0xf93f87b7442e45d4,
+      0xafebff0bcb24aafe, 0xf78f69a51539d749,
+      0xdbe6fecebdedd5be, 0xb573440e5a884d1c,
+      0x89705f4136b4a597, 0x31680a88f8953031,
+      0xabcc77118461cefc, 0xfdc20d2b36ba7c3e,
+      0xd6bf94d5e57a42bc, 0x3d32907604691b4d,
+      0x8637bd05af6c69b5, 0xa63f9a49c2c1b110,
+      0xa7c5ac471b478423, 0xfcf80dc33721d54,
+      0xd1b71758e219652b, 0xd3c36113404ea4a9,
+      0x83126e978d4fdf3b, 0x645a1cac083126ea,
+      0xa3d70a3d70a3d70a, 0x3d70a3d70a3d70a4,
+      0xcccccccccccccccc, 0xcccccccccccccccd,
+      0x8000000000000000, 0x0,
+      0xa000000000000000, 0x0,
+      0xc800000000000000, 0x0,
+      0xfa00000000000000, 0x0,
+      0x9c40000000000000, 0x0,
+      0xc350000000000000, 0x0,
+      0xf424000000000000, 0x0,
+      0x9896800000000000, 0x0,
+      0xbebc200000000000, 0x0,
+      0xee6b280000000000, 0x0,
+      0x9502f90000000000, 0x0,
+      0xba43b74000000000, 0x0,
+      0xe8d4a51000000000, 0x0,
+      0x9184e72a00000000, 0x0,
+      0xb5e620f480000000, 0x0,
+      0xe35fa931a0000000, 0x0,
+      0x8e1bc9bf04000000, 0x0,
+      0xb1a2bc2ec5000000, 0x0,
+      0xde0b6b3a76400000, 0x0,
+      0x8ac7230489e80000, 0x0,
+      0xad78ebc5ac620000, 0x0,
+      0xd8d726b7177a8000, 0x0,
+      0x878678326eac9000, 0x0,
+      0xa968163f0a57b400, 0x0,
+      0xd3c21bcecceda100, 0x0,
+      0x84595161401484a0, 0x0,
+      0xa56fa5b99019a5c8, 0x0,
+      0xcecb8f27f4200f3a, 0x0,
+      0x813f3978f8940984, 0x4000000000000000,
+      0xa18f07d736b90be5, 0x5000000000000000,
+      0xc9f2c9cd04674ede, 0xa400000000000000,
+      0xfc6f7c4045812296, 0x4d00000000000000,
+      0x9dc5ada82b70b59d, 0xf020000000000000,
+      0xc5371912364ce305, 0x6c28000000000000,
+      0xf684df56c3e01bc6, 0xc732000000000000,
+      0x9a130b963a6c115c, 0x3c7f400000000000,
+      0xc097ce7bc90715b3, 0x4b9f100000000000,
+      0xf0bdc21abb48db20, 0x1e86d40000000000,
+      0x96769950b50d88f4, 0x1314448000000000,
+      0xbc143fa4e250eb31, 0x17d955a000000000,
+      0xeb194f8e1ae525fd, 0x5dcfab0800000000,
+      0x92efd1b8d0cf37be, 0x5aa1cae500000000,
+      0xb7abc627050305ad, 0xf14a3d9e40000000,
+      0xe596b7b0c643c719, 0x6d9ccd05d0000000,
+      0x8f7e32ce7bea5c6f, 0xe4820023a2000000,
+      0xb35dbf821ae4f38b, 0xdda2802c8a800000,
+      0xe0352f62a19e306e, 0xd50b2037ad200000,
+      0x8c213d9da502de45, 0x4526f422cc340000,
+      0xaf298d050e4395d6, 0x9670b12b7f410000,
+      0xdaf3f04651d47b4c, 0x3c0cdd765f114000,
+      0x88d8762bf324cd0f, 0xa5880a69fb6ac800,
+      0xab0e93b6efee0053, 0x8eea0d047a457a00,
+      0xd5d238a4abe98068, 0x72a4904598d6d880,
+      0x85a36366eb71f041, 0x47a6da2b7f864750,
+      0xa70c3c40a64e6c51, 0x999090b65f67d924,
+      0xd0cf4b50cfe20765, 0xfff4b4e3f741cf6d,
+      0x82818f1281ed449f, 0xbff8f10e7a8921a4,
+      0xa321f2d7226895c7, 0xaff72d52192b6a0d,
+      0xcbea6f8ceb02bb39, 0x9bf4f8a69f764490,
+      0xfee50b7025c36a08, 0x2f236d04753d5b4,
+      0x9f4f2726179a2245, 0x1d762422c946590,
+      0xc722f0ef9d80aad6, 0x424d3ad2b7b97ef5,
+      0xf8ebad2b84e0d58b, 0xd2e0898765a7deb2,
+      0x9b934c3b330c8577, 0x63cc55f49f88eb2f,
+      0xc2781f49ffcfa6d5, 0x3cbf6b71c76b25fb,
+      0xf316271c7fc3908a, 0x8bef464e3945ef7a,
+      0x97edd871cfda3a56, 0x97758bf0e3cbb5ac,
+      0xbde94e8e43d0c8ec, 0x3d52eeed1cbea317,
+      0xed63a231d4c4fb27, 0x4ca7aaa863ee4bdd,
+      0x945e455f24fb1cf8, 0x8fe8caa93e74ef6a,
+      0xb975d6b6ee39e436, 0xb3e2fd538e122b44,
+      0xe7d34c64a9c85d44, 0x60dbbca87196b616,
+      0x90e40fbeea1d3a4a, 0xbc8955e946fe31cd,
+      0xb51d13aea4a488dd, 0x6babab6398bdbe41,
+      0xe264589a4dcdab14, 0xc696963c7eed2dd1,
+      0x8d7eb76070a08aec, 0xfc1e1de5cf543ca2,
+      0xb0de65388cc8ada8, 0x3b25a55f43294bcb,
+      0xdd15fe86affad912, 0x49ef0eb713f39ebe,
+      0x8a2dbf142dfcc7ab, 0x6e3569326c784337,
+      0xacb92ed9397bf996, 0x49c2c37f07965404,
+      0xd7e77a8f87daf7fb, 0xdc33745ec97be906,
+      0x86f0ac99b4e8dafd, 0x69a028bb3ded71a3,
+      0xa8acd7c0222311bc, 0xc40832ea0d68ce0c,
+      0xd2d80db02aabd62b, 0xf50a3fa490c30190,
+      0x83c7088e1aab65db, 0x792667c6da79e0fa,
+      0xa4b8cab1a1563f52, 0x577001b891185938,
+      0xcde6fd5e09abcf26, 0xed4c0226b55e6f86,
+      0x80b05e5ac60b6178, 0x544f8158315b05b4,
+      0xa0dc75f1778e39d6, 0x696361ae3db1c721,
+      0xc913936dd571c84c, 0x3bc3a19cd1e38e9,
+      0xfb5878494ace3a5f, 0x4ab48a04065c723,
+      0x9d174b2dcec0e47b, 0x62eb0d64283f9c76,
+      0xc45d1df942711d9a, 0x3ba5d0bd324f8394,
+      0xf5746577930d6500, 0xca8f44ec7ee36479,
+      0x9968bf6abbe85f20, 0x7e998b13cf4e1ecb,
+      0xbfc2ef456ae276e8, 0x9e3fedd8c321a67e,
+      0xefb3ab16c59b14a2, 0xc5cfe94ef3ea101e,
+      0x95d04aee3b80ece5, 0xbba1f1d158724a12,
+      0xbb445da9ca61281f, 0x2a8a6e45ae8edc97,
+      0xea1575143cf97226, 0xf52d09d71a3293bd,
+      0x924d692ca61be758, 0x593c2626705f9c56,
+      0xb6e0c377cfa2e12e, 0x6f8b2fb00c77836c,
+      0xe498f455c38b997a, 0xb6dfb9c0f956447,
+      0x8edf98b59a373fec, 0x4724bd4189bd5eac,
+      0xb2977ee300c50fe7, 0x58edec91ec2cb657,
+      0xdf3d5e9bc0f653e1, 0x2f2967b66737e3ed,
+      0x8b865b215899f46c, 0xbd79e0d20082ee74,
+      0xae67f1e9aec07187, 0xecd8590680a3aa11,
+      0xda01ee641a708de9, 0xe80e6f4820cc9495,
+      0x884134fe908658b2, 0x3109058d147fdcdd,
+      0xaa51823e34a7eede, 0xbd4b46f0599fd415,
+      0xd4e5e2cdc1d1ea96, 0x6c9e18ac7007c91a,
+      0x850fadc09923329e, 0x3e2cf6bc604ddb0,
+      0xa6539930bf6bff45, 0x84db8346b786151c,
+      0xcfe87f7cef46ff16, 0xe612641865679a63,
+      0x81f14fae158c5f6e, 0x4fcb7e8f3f60c07e,
+      0xa26da3999aef7749, 0xe3be5e330f38f09d,
+      0xcb090c8001ab551c, 0x5cadf5bfd3072cc5,
+      0xfdcb4fa002162a63, 0x73d9732fc7c8f7f6,
+      0x9e9f11c4014dda7e, 0x2867e7fddcdd9afa,
+      0xc646d63501a1511d, 0xb281e1fd541501b8,
+      0xf7d88bc24209a565, 0x1f225a7ca91a4226,
+      0x9ae757596946075f, 0x3375788de9b06958,
+      0xc1a12d2fc3978937, 0x52d6b1641c83ae,
+      0xf209787bb47d6b84, 0xc0678c5dbd23a49a,
+      0x9745eb4d50ce6332, 0xf840b7ba963646e0,
+      0xbd176620a501fbff, 0xb650e5a93bc3d898,
+      0xec5d3fa8ce427aff, 0xa3e51f138ab4cebe,
+      0x93ba47c980e98cdf, 0xc66f336c36b10137,
+      0xb8a8d9bbe123f017, 0xb80b0047445d4184,
+      0xe6d3102ad96cec1d, 0xa60dc059157491e5,
+      0x9043ea1ac7e41392, 0x87c89837ad68db2f,
+      0xb454e4a179dd1877, 0x29babe4598c311fb,
+      0xe16a1dc9d8545e94, 0xf4296dd6fef3d67a,
+      0x8ce2529e2734bb1d, 0x1899e4a65f58660c,
+      0xb01ae745b101e9e4, 0x5ec05dcff72e7f8f,
+      0xdc21a1171d42645d, 0x76707543f4fa1f73,
+      0x899504ae72497eba, 0x6a06494a791c53a8,
+      0xabfa45da0edbde69, 0x487db9d17636892,
+      0xd6f8d7509292d603, 0x45a9d2845d3c42b6,
+      0x865b86925b9bc5c2, 0xb8a2392ba45a9b2,
+      0xa7f26836f282b732, 0x8e6cac7768d7141e,
+      0xd1ef0244af2364ff, 0x3207d795430cd926,
+      0x8335616aed761f1f, 0x7f44e6bd49e807b8,
+      0xa402b9c5a8d3a6e7, 0x5f16206c9c6209a6,
+      0xcd036837130890a1, 0x36dba887c37a8c0f,
+      0x802221226be55a64, 0xc2494954da2c9789,
+      0xa02aa96b06deb0fd, 0xf2db9baa10b7bd6c,
+      0xc83553c5c8965d3d, 0x6f92829494e5acc7,
+      0xfa42a8b73abbf48c, 0xcb772339ba1f17f9,
+      0x9c69a97284b578d7, 0xff2a760414536efb,
+      0xc38413cf25e2d70d, 0xfef5138519684aba,
+      0xf46518c2ef5b8cd1, 0x7eb258665fc25d69,
+      0x98bf2f79d5993802, 0xef2f773ffbd97a61,
+      0xbeeefb584aff8603, 0xaafb550ffacfd8fa,
+      0xeeaaba2e5dbf6784, 0x95ba2a53f983cf38,
+      0x952ab45cfa97a0b2, 0xdd945a747bf26183,
+      0xba756174393d88df, 0x94f971119aeef9e4,
+      0xe912b9d1478ceb17, 0x7a37cd5601aab85d,
+      0x91abb422ccb812ee, 0xac62e055c10ab33a,
+      0xb616a12b7fe617aa, 0x577b986b314d6009,
+      0xe39c49765fdf9d94, 0xed5a7e85fda0b80b,
+      0x8e41ade9fbebc27d, 0x14588f13be847307,
+      0xb1d219647ae6b31c, 0x596eb2d8ae258fc8,
+      0xde469fbd99a05fe3, 0x6fca5f8ed9aef3bb,
+      0x8aec23d680043bee, 0x25de7bb9480d5854,
+      0xada72ccc20054ae9, 0xaf561aa79a10ae6a,
+      0xd910f7ff28069da4, 0x1b2ba1518094da04,
+      0x87aa9aff79042286, 0x90fb44d2f05d0842,
+      0xa99541bf57452b28, 0x353a1607ac744a53,
+      0xd3fa922f2d1675f2, 0x42889b8997915ce8,
+      0x847c9b5d7c2e09b7, 0x69956135febada11,
+      0xa59bc234db398c25, 0x43fab9837e699095,
+      0xcf02b2c21207ef2e, 0x94f967e45e03f4bb,
+      0x8161afb94b44f57d, 0x1d1be0eebac278f5,
+      0xa1ba1ba79e1632dc, 0x6462d92a69731732,
+      0xca28a291859bbf93, 0x7d7b8f7503cfdcfe,
+      0xfcb2cb35e702af78, 0x5cda735244c3d43e,
+      0x9defbf01b061adab, 0x3a0888136afa64a7,
+      0xc56baec21c7a1916, 0x88aaa1845b8fdd0,
+      0xf6c69a72a3989f5b, 0x8aad549e57273d45,
+      0x9a3c2087a63f6399, 0x36ac54e2f678864b,
+      0xc0cb28a98fcf3c7f, 0x84576a1bb416a7dd,
+      0xf0fdf2d3f3c30b9f, 0x656d44a2a11c51d5,
+      0x969eb7c47859e743, 0x9f644ae5a4b1b325,
+      0xbc4665b596706114, 0x873d5d9f0dde1fee,
+      0xeb57ff22fc0c7959, 0xa90cb506d155a7ea,
+      0x9316ff75dd87cbd8, 0x9a7f12442d588f2,
+      0xb7dcbf5354e9bece, 0xc11ed6d538aeb2f,
+      0xe5d3ef282a242e81, 0x8f1668c8a86da5fa,
+      0x8fa475791a569d10, 0xf96e017d694487bc,
+      0xb38d92d760ec4455, 0x37c981dcc395a9ac,
+      0xe070f78d3927556a, 0x85bbe253f47b1417,
+      0x8c469ab843b89562, 0x93956d7478ccec8e,
+      0xaf58416654a6babb, 0x387ac8d1970027b2,
+      0xdb2e51bfe9d0696a, 0x6997b05fcc0319e,
+      0x88fcf317f22241e2, 0x441fece3bdf81f03,
+      0xab3c2fddeeaad25a, 0xd527e81cad7626c3,
+      0xd60b3bd56a5586f1, 0x8a71e223d8d3b074,
+      0x85c7056562757456, 0xf6872d5667844e49,
+      0xa738c6bebb12d16c, 0xb428f8ac016561db,
+      0xd106f86e69d785c7, 0xe13336d701beba52,
+      0x82a45b450226b39c, 0xecc0024661173473,
+      0xa34d721642b06084, 0x27f002d7f95d0190,
+      0xcc20ce9bd35c78a5, 0x31ec038df7b441f4,
+      0xff290242c83396ce, 0x7e67047175a15271,
+      0x9f79a169bd203e41, 0xf0062c6e984d386,
+      0xc75809c42c684dd1, 0x52c07b78a3e60868,
+      0xf92e0c3537826145, 0xa7709a56ccdf8a82,
+      0x9bbcc7a142b17ccb, 0x88a66076400bb691,
+      0xc2abf989935ddbfe, 0x6acff893d00ea435,
+      0xf356f7ebf83552fe, 0x583f6b8c4124d43,
+      0x98165af37b2153de, 0xc3727a337a8b704a,
+      0xbe1bf1b059e9a8d6, 0x744f18c0592e4c5c,
+      0xeda2ee1c7064130c, 0x1162def06f79df73,
+      0x9485d4d1c63e8be7, 0x8addcb5645ac2ba8,
+      0xb9a74a0637ce2ee1, 0x6d953e2bd7173692,
+      0xe8111c87c5c1ba99, 0xc8fa8db6ccdd0437,
+      0x910ab1d4db9914a0, 0x1d9c9892400a22a2,
+      0xb54d5e4a127f59c8, 0x2503beb6d00cab4b,
+      0xe2a0b5dc971f303a, 0x2e44ae64840fd61d,
+      0x8da471a9de737e24, 0x5ceaecfed289e5d2,
+      0xb10d8e1456105dad, 0x7425a83e872c5f47,
+      0xdd50f1996b947518, 0xd12f124e28f77719,
+      0x8a5296ffe33cc92f, 0x82bd6b70d99aaa6f,
+      0xace73cbfdc0bfb7b, 0x636cc64d1001550b,
+      0xd8210befd30efa5a, 0x3c47f7e05401aa4e,
+      0x8714a775e3e95c78, 0x65acfaec34810a71,
+      0xa8d9d1535ce3b396, 0x7f1839a741a14d0d,
+      0xd31045a8341ca07c, 0x1ede48111209a050,
+      0x83ea2b892091e44d, 0x934aed0aab460432,
+      0xa4e4b66b68b65d60, 0xf81da84d5617853f,
+      0xce1de40642e3f4b9, 0x36251260ab9d668e,
+      0x80d2ae83e9ce78f3, 0xc1d72b7c6b426019,
+      0xa1075a24e4421730, 0xb24cf65b8612f81f,
+      0xc94930ae1d529cfc, 0xdee033f26797b627,
+      0xfb9b7cd9a4a7443c, 0x169840ef017da3b1,
+      0x9d412e0806e88aa5, 0x8e1f289560ee864e,
+      0xc491798a08a2ad4e, 0xf1a6f2bab92a27e2,
+      0xf5b5d7ec8acb58a2, 0xae10af696774b1db,
+      0x9991a6f3d6bf1765, 0xacca6da1e0a8ef29,
+      0xbff610b0cc6edd3f, 0x17fd090a58d32af3,
+      0xeff394dcff8a948e, 0xddfc4b4cef07f5b0,
+      0x95f83d0a1fb69cd9, 0x4abdaf101564f98e,
+      0xbb764c4ca7a4440f, 0x9d6d1ad41abe37f1,
+      0xea53df5fd18d5513, 0x84c86189216dc5ed,
+      0x92746b9be2f8552c, 0x32fd3cf5b4e49bb4,
+      0xb7118682dbb66a77, 0x3fbc8c33221dc2a1,
+      0xe4d5e82392a40515, 0xfabaf3feaa5334a,
+      0x8f05b1163ba6832d, 0x29cb4d87f2a7400e,
+      0xb2c71d5bca9023f8, 0x743e20e9ef511012,
+      0xdf78e4b2bd342cf6, 0x914da9246b255416,
+      0x8bab8eefb6409c1a, 0x1ad089b6c2f7548e,
+      0xae9672aba3d0c320, 0xa184ac2473b529b1,
+      0xda3c0f568cc4f3e8, 0xc9e5d72d90a2741e,
+      0x8865899617fb1871, 0x7e2fa67c7a658892,
+      0xaa7eebfb9df9de8d, 0xddbb901b98feeab7,
+      0xd51ea6fa85785631, 0x552a74227f3ea565,
+      0x8533285c936b35de, 0xd53a88958f87275f,
+      0xa67ff273b8460356, 0x8a892abaf368f137,
+      0xd01fef10a657842c, 0x2d2b7569b0432d85,
+      0x8213f56a67f6b29b, 0x9c3b29620e29fc73,
+      0xa298f2c501f45f42, 0x8349f3ba91b47b8f,
+      0xcb3f2f7642717713, 0x241c70a936219a73,
+      0xfe0efb53d30dd4d7, 0xed238cd383aa0110,
+      0x9ec95d1463e8a506, 0xf4363804324a40aa,
+      0xc67bb4597ce2ce48, 0xb143c6053edcd0d5,
+      0xf81aa16fdc1b81da, 0xdd94b7868e94050a,
+      0x9b10a4e5e9913128, 0xca7cf2b4191c8326,
+      0xc1d4ce1f63f57d72, 0xfd1c2f611f63a3f0,
+      0xf24a01a73cf2dccf, 0xbc633b39673c8cec,
+      0x976e41088617ca01, 0xd5be0503e085d813,
+      0xbd49d14aa79dbc82, 0x4b2d8644d8a74e18,
+      0xec9c459d51852ba2, 0xddf8e7d60ed1219e,
+      0x93e1ab8252f33b45, 0xcabb90e5c942b503,
+      0xb8da1662e7b00a17, 0x3d6a751f3b936243,
+      0xe7109bfba19c0c9d, 0xcc512670a783ad4,
+      0x906a617d450187e2, 0x27fb2b80668b24c5,
+      0xb484f9dc9641e9da, 0xb1f9f660802dedf6,
+      0xe1a63853bbd26451, 0x5e7873f8a0396973,
+      0x8d07e33455637eb2, 0xdb0b487b6423e1e8,
+      0xb049dc016abc5e5f, 0x91ce1a9a3d2cda62,
+      0xdc5c5301c56b75f7, 0x7641a140cc7810fb,
+      0x89b9b3e11b6329ba, 0xa9e904c87fcb0a9d,
+      0xac2820d9623bf429, 0x546345fa9fbdcd44,
+      0xd732290fbacaf133, 0xa97c177947ad4095,
+      0x867f59a9d4bed6c0, 0x49ed8eabcccc485d,
+      0xa81f301449ee8c70, 0x5c68f256bfff5a74,
+      0xd226fc195c6a2f8c, 0x73832eec6fff3111,
+      0x83585d8fd9c25db7, 0xc831fd53c5ff7eab,
+      0xa42e74f3d032f525, 0xba3e7ca8b77f5e55,
+      0xcd3a1230c43fb26f, 0x28ce1bd2e55f35eb,
+      0x80444b5e7aa7cf85, 0x7980d163cf5b81b3,
+      0xa0555e361951c366, 0xd7e105bcc332621f,
+      0xc86ab5c39fa63440, 0x8dd9472bf3fefaa7,
+      0xfa856334878fc150, 0xb14f98f6f0feb951,
+      0x9c935e00d4b9d8d2, 0x6ed1bf9a569f33d3,
+      0xc3b8358109e84f07, 0xa862f80ec4700c8,
+      0xf4a642e14c6262c8, 0xcd27bb612758c0fa,
+      0x98e7e9cccfbd7dbd, 0x8038d51cb897789c,
+      0xbf21e44003acdd2c, 0xe0470a63e6bd56c3,
+      0xeeea5d5004981478, 0x1858ccfce06cac74,
+      0x95527a5202df0ccb, 0xf37801e0c43ebc8,
+      0xbaa718e68396cffd, 0xd30560258f54e6ba,
+      0xe950df20247c83fd, 0x47c6b82ef32a2069,
+      0x91d28b7416cdd27e, 0x4cdc331d57fa5441,
+      0xb6472e511c81471d, 0xe0133fe4adf8e952,
+      0xe3d8f9e563a198e5, 0x58180fddd97723a6,
+      0x8e679c2f5e44ff8f, 0x570f09eaa7ea7648,
+  };
+};
+
+#if SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE
+
+template <class unused>
+constexpr uint64_t
+    powers_template<unused>::power_of_five_128[number_of_entries];
+
+#endif
+
+using powers = powers_template<>;
+
+} // namespace simdjson_fast_float
+
+#endif
+
+#ifndef SIMDJSON_FASTFLOAT_DECIMAL_TO_BINARY_H
+#define SIMDJSON_FASTFLOAT_DECIMAL_TO_BINARY_H
+
+#include <cfloat>
+#include <cinttypes>
+#include <cmath>
+#include <cstdint>
+#include <cstdlib>
+#include <cstring>
+
+namespace simdjson_fast_float {
+
+// This will compute or rather approximate w * 5**q and return a pair of 64-bit
+// words approximating the result, with the "high" part corresponding to the
+// most significant bits and the low part corresponding to the least significant
+// bits.
+//
+template <int bit_precision>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 value128
+compute_product_approximation(int64_t q, uint64_t w) {
+  int const index = 2 * int(q - powers::smallest_power_of_five);
+  // For small values of q, e.g., q in [0,27], the answer is always exact
+  // because The line value128 firstproduct = full_multiplication(w,
+  // power_of_five_128[index]); gives the exact answer.
+  value128 firstproduct =
+      full_multiplication(w, powers::power_of_five_128[index]);
+  static_assert((bit_precision >= 0) && (bit_precision <= 64),
+                " precision should  be in (0,64]");
+  constexpr uint64_t precision_mask =
+      (bit_precision < 64) ? (uint64_t(0xFFFFFFFFFFFFFFFF) >> bit_precision)
+                           : uint64_t(0xFFFFFFFFFFFFFFFF);
+  if ((firstproduct.high & precision_mask) ==
+      precision_mask) { // could further guard with  (lower + w < lower)
+    // regarding the second product, we only need secondproduct.high, but our
+    // expectation is that the compiler will optimize this extra work away if
+    // needed.
+    value128 secondproduct =
+        full_multiplication(w, powers::power_of_five_128[index + 1]);
     firstproduct.low += secondproduct.high;
     if (secondproduct.high > firstproduct.low) {
       firstproduct.high++;
     }
   }
-  uint64_t lower = firstproduct.low;
-  uint64_t upper = firstproduct.high;
-  uint64_t upperbit = upper >> 63;
-  uint64_t mantissa = upper >> (upperbit + 9);
-  lz += int(1 ^ upperbit);
-  int64_t real_exponent = exponent - lz;
-  if (real_exponent <= 0) {
-    if (-real_exponent + 1 >= 64) {
-      d = negative ? -0.0 : 0.0;
+  return firstproduct;
+}
+
+namespace detail {
+/**
+ * For q in (0,350), we have that
+ *  f = (((152170 + 65536) * q ) >> 16);
+ * is equal to
+ *   floor(p) + q
+ * where
+ *   p = log(5**q)/log(2) = q * log(5)/log(2)
+ *
+ * For negative values of q in (-400,0), we have that
+ *  f = (((152170 + 65536) * q ) >> 16);
+ * is equal to
+ *   -ceil(p) + q
+ * where
+ *   p = log(5**-q)/log(2) = -q * log(5)/log(2)
+ */
+constexpr simdjson_fastfloat_really_inline int32_t power(int32_t q) noexcept {
+  return (((152170 + 65536) * q) >> 16) + 63;
+}
+} // namespace detail
+
+// create an adjusted mantissa, biased by the invalid power2
+// for significant digits already multiplied by 10 ** q.
+template <typename binary>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 adjusted_mantissa
+compute_error_scaled(int64_t q, uint64_t w, int lz) noexcept {
+  int hilz = int(w >> 63) ^ 1;
+  adjusted_mantissa answer;
+  answer.mantissa = w << hilz;
+  int bias = binary::mantissa_explicit_bits() - binary::minimum_exponent();
+  answer.power2 = int32_t(detail::power(int32_t(q)) + bias - hilz - lz - 62 +
+                          invalid_am_bias);
+  return answer;
+}
+
+// w * 10 ** q, without rounding the representation up.
+// the power2 in the exponent will be adjusted by invalid_am_bias.
+template <typename binary>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 adjusted_mantissa
+compute_error(int64_t q, uint64_t w) noexcept {
+  int lz = leading_zeroes(w);
+  w <<= lz;
+  value128 product =
+      compute_product_approximation<binary::mantissa_explicit_bits() + 3>(q, w);
+  return compute_error_scaled<binary>(q, product.high, lz);
+}
+
+// Computers w * 10 ** q.
+// The returned value should be a valid number that simply needs to be
+// packed. However, in some very rare cases, the computation will fail. In such
+// cases, we return an adjusted_mantissa with a negative power of 2: the caller
+// should recompute in such cases.
+template <typename binary>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 adjusted_mantissa
+compute_float(int64_t q, uint64_t w) noexcept {
+  adjusted_mantissa answer;
+  if ((w == 0) || (q < binary::smallest_power_of_ten())) {
+    answer.power2 = 0;
+    answer.mantissa = 0;
+    // result should be zero
+    return answer;
+  }
+  if (q > binary::largest_power_of_ten()) {
+    // we want to get infinity:
+    answer.power2 = binary::infinite_power();
+    answer.mantissa = 0;
+    return answer;
+  }
+  // At this point in time q is in [powers::smallest_power_of_five,
+  // powers::largest_power_of_five].
+
+  // We want the most significant bit of i to be 1. Shift if needed.
+  int lz = leading_zeroes(w);
+  w <<= lz;
+
+  // The required precision is binary::mantissa_explicit_bits() + 3 because
+  // 1. We need the implicit bit
+  // 2. We need an extra bit for rounding purposes
+  // 3. We might lose a bit due to the "upperbit" routine (result too small,
+  // requiring a shift)
+
+  value128 product =
+      compute_product_approximation<binary::mantissa_explicit_bits() + 3>(q, w);
+  // The computed 'product' is always sufficient.
+  // Mathematical proof:
+  // Noble Mushtak and Daniel Lemire, Fast Number Parsing Without Fallback (to
+  // appear) See script/mushtak_lemire.py
+
+  // The "compute_product_approximation" function can be slightly slower than a
+  // branchless approach: value128 product = compute_product(q, w); but in
+  // practice, we can win big with the compute_product_approximation if its
+  // additional branch is easily predicted. Which is best is data specific.
+  int upperbit = int(product.high >> 63);
+  int shift = upperbit + 64 - binary::mantissa_explicit_bits() - 3;
+
+  answer.mantissa = product.high >> shift;
+
+  answer.power2 = int32_t(detail::power(int32_t(q)) + upperbit - lz -
+                          binary::minimum_exponent());
+  if (answer.power2 <= 0) { // we have a subnormal?
+    // Here have that answer.power2 <= 0 so -answer.power2 >= 0
+    if (-answer.power2 + 1 >=
+        64) { // if we have more than 64 bits below the minimum exponent, you
+              // have a zero for sure.
+      answer.power2 = 0;
+      answer.mantissa = 0;
+      // result should be zero
+      return answer;
+    }
+    // next line is safe because -answer.power2 + 1 < 64
+    answer.mantissa >>= -answer.power2 + 1;
+    // Thankfully, we can't have both "round-to-even" and subnormals because
+    // "round-to-even" only occurs for powers close to 0 in the 32-bit and
+    // and 64-bit case (with no more than 19 digits).
+    answer.mantissa += (answer.mantissa & 1); // round up
+    answer.mantissa >>= 1;
+    // There is a weird scenario where we don't have a subnormal but just.
+    // Suppose we start with 2.2250738585072013e-308, we end up
+    // with 0x3fffffffffffff x 2^-1023-53 which is technically subnormal
+    // whereas 0x40000000000000 x 2^-1023-53  is normal. Now, we need to round
+    // up 0x3fffffffffffff x 2^-1023-53  and once we do, we are no longer
+    // subnormal, but we can only know this after rounding.
+    // So we only declare a subnormal if we are smaller than the threshold.
+    answer.power2 =
+        (answer.mantissa < (uint64_t(1) << binary::mantissa_explicit_bits()))
+            ? 0
+            : 1;
+    return answer;
+  }
+
+  // usually, we round *up*, but if we fall right in between and and we have an
+  // even basis, we need to round down
+  // We are only concerned with the cases where 5**q fits in single 64-bit word.
+  if ((product.low <= 1) && (q >= binary::min_exponent_round_to_even()) &&
+      (q <= binary::max_exponent_round_to_even()) &&
+      ((answer.mantissa & 3) == 1)) { // we may fall between two floats!
+    // To be in-between two floats we need that in doing
+    //   answer.mantissa = product.high >> (upperbit + 64 -
+    //   binary::mantissa_explicit_bits() - 3);
+    // ... we dropped out only zeroes. But if this happened, then we can go
+    // back!!!
+    if ((answer.mantissa << shift) == product.high) {
+      answer.mantissa &= ~uint64_t(1); // flip it so that we do not round up
+    }
+  }
+
+  answer.mantissa += (answer.mantissa & 1); // round up
+  answer.mantissa >>= 1;
+  if (answer.mantissa >= (uint64_t(2) << binary::mantissa_explicit_bits())) {
+    answer.mantissa = (uint64_t(1) << binary::mantissa_explicit_bits());
+    answer.power2++; // undo previous addition
+  }
+
+  answer.mantissa &= ~(uint64_t(1) << binary::mantissa_explicit_bits());
+  if (answer.power2 >= binary::infinite_power()) { // infinity
+    answer.power2 = binary::infinite_power();
+    answer.mantissa = 0;
+  }
+  return answer;
+}
+
+} // namespace simdjson_fast_float
+
+#endif
+
+#ifndef SIMDJSON_FASTFLOAT_BIGINT_H
+#define SIMDJSON_FASTFLOAT_BIGINT_H
+
+#include <algorithm>
+#include <cstdint>
+#include <climits>
+#include <cstring>
+
+
+namespace simdjson_fast_float {
+
+// the limb width: we want efficient multiplication of double the bits in
+// limb, or for 64-bit limbs, at least 64-bit multiplication where we can
+// extract the high and low parts efficiently. this is every 64-bit
+// architecture except for sparc, which emulates 128-bit multiplication.
+// we might have platforms where `CHAR_BIT` is not 8, so let's avoid
+// doing `8 * sizeof(limb)`.
+#if defined(SIMDJSON_FASTFLOAT_64BIT) && !defined(__sparc)
+#define SIMDJSON_FASTFLOAT_64BIT_LIMB 1
+typedef uint64_t limb;
+constexpr size_t limb_bits = 64;
+#else
+#define SIMDJSON_FASTFLOAT_32BIT_LIMB
+typedef uint32_t limb;
+constexpr size_t limb_bits = 32;
+#endif
+
+typedef span<limb> limb_span;
+
+// number of bits in a bigint. this needs to be at least the number
+// of bits required to store the largest bigint, which is
+// `log2(10**(digits + max_exp))`, or `log2(10**(767 + 342))`, or
+// ~3600 bits, so we round to 4000.
+constexpr size_t bigint_bits = 4000;
+constexpr size_t bigint_limbs = bigint_bits / limb_bits;
+
+// vector-like type that is allocated on the stack. the entire
+// buffer is pre-allocated, and only the length changes.
+template <uint16_t size> struct stackvec {
+  limb data[size];
+  // we never need more than 150 limbs
+  uint16_t length{0};
+
+  stackvec() = default;
+  stackvec(stackvec const &) = delete;
+  stackvec &operator=(stackvec const &) = delete;
+  stackvec(stackvec &&) = delete;
+  stackvec &operator=(stackvec &&other) = delete;
+
+  // create stack vector from existing limb span.
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 stackvec(limb_span s) {
+    SIMDJSON_FASTFLOAT_ASSERT(try_extend(s));
+  }
+
+  SIMDJSON_FASTFLOAT_CONSTEXPR14 limb &operator[](size_t index) noexcept {
+    SIMDJSON_FASTFLOAT_DEBUG_ASSERT(index < length);
+    return data[index];
+  }
+
+  SIMDJSON_FASTFLOAT_CONSTEXPR14 const limb &operator[](size_t index) const noexcept {
+    SIMDJSON_FASTFLOAT_DEBUG_ASSERT(index < length);
+    return data[index];
+  }
+
+  // index from the end of the container
+  SIMDJSON_FASTFLOAT_CONSTEXPR14 const limb &rindex(size_t index) const noexcept {
+    SIMDJSON_FASTFLOAT_DEBUG_ASSERT(index < length);
+    size_t rindex = length - index - 1;
+    return data[rindex];
+  }
+
+  // set the length, without bounds checking.
+  SIMDJSON_FASTFLOAT_CONSTEXPR14 void set_len(size_t len) noexcept {
+    length = uint16_t(len);
+  }
+
+  constexpr size_t len() const noexcept { return length; }
+
+  constexpr bool is_empty() const noexcept { return length == 0; }
+
+  constexpr size_t capacity() const noexcept { return size; }
+
+  // append item to vector, without bounds checking
+  SIMDJSON_FASTFLOAT_CONSTEXPR14 void push_unchecked(limb value) noexcept {
+    data[length] = value;
+    length++;
+  }
+
+  // append item to vector, returning if item was added
+  SIMDJSON_FASTFLOAT_CONSTEXPR14 bool try_push(limb value) noexcept {
+    if (len() < capacity()) {
+      push_unchecked(value);
       return true;
+    } else {
+      return false;
     }
-    mantissa >>= -real_exponent + 1;
-    mantissa += (mantissa & 1);
-    mantissa >>= 1;
-    real_exponent = (mantissa < (uint64_t(1) << 52)) ? 0 : 1;
-    d = to_double(mantissa, real_exponent, negative);
-    return true;
   }
-  if ((lower <= 1) && (power >= -4) && (power <= 23) && ((mantissa & 3) == 1)) {
-    if ((mantissa << (upperbit + 64 - 53 - 2)) == upper) {
-      mantissa &= ~1;
+
+  // add items to the vector, from a span, without bounds checking
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 void extend_unchecked(limb_span s) noexcept {
+    limb *ptr = data + length;
+    std::copy_n(s.ptr, s.len(), ptr);
+    set_len(len() + s.len());
+  }
+
+  // try to add items to the vector, returning if items were added
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 bool try_extend(limb_span s) noexcept {
+    if (len() + s.len() <= capacity()) {
+      extend_unchecked(s);
+      return true;
+    } else {
+      return false;
     }
   }
-  mantissa += mantissa & 1;
-  mantissa >>= 1;
-  if (mantissa >= (1ULL << 53)) {
-    mantissa = (1ULL << 52);
-    real_exponent++;
+
+  // resize the vector, without bounds checking
+  // if the new size is longer than the vector, assign value to each
+  // appended item.
+  SIMDJSON_FASTFLOAT_CONSTEXPR20
+  void resize_unchecked(size_t new_len, limb value) noexcept {
+    if (new_len > len()) {
+      size_t count = new_len - len();
+      limb *first = data + len();
+      limb *last = first + count;
+      ::std::fill(first, last, value);
+      set_len(new_len);
+    } else {
+      set_len(new_len);
+    }
   }
-  mantissa &= ~(1ULL << 52);
-  if (real_exponent > 2046) {
+
+  // try to resize the vector, returning if the vector was resized.
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 bool try_resize(size_t new_len, limb value) noexcept {
+    if (new_len > capacity()) {
+      return false;
+    } else {
+      resize_unchecked(new_len, value);
+      return true;
+    }
+  }
+
+  // check if any limbs are non-zero after the given index.
+  // this needs to be done in reverse order, since the index
+  // is relative to the most significant limbs.
+  SIMDJSON_FASTFLOAT_CONSTEXPR14 bool nonzero(size_t index) const noexcept {
+    while (index < len()) {
+      if (rindex(index) != 0) {
+        return true;
+      }
+      index++;
+    }
     return false;
   }
-  d = to_double(mantissa, real_exponent, negative);
+
+  // normalize the big integer, so most-significant zero limbs are removed.
+  SIMDJSON_FASTFLOAT_CONSTEXPR14 void normalize() noexcept {
+    while (len() > 0 && rindex(0) == 0) {
+      length--;
+    }
+  }
+};
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 uint64_t
+empty_hi64(bool &truncated) noexcept {
+  truncated = false;
+  return 0;
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 uint64_t
+uint64_hi64(uint64_t r0, bool &truncated) noexcept {
+  truncated = false;
+  int shl = leading_zeroes(r0);
+  return r0 << shl;
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 uint64_t
+uint64_hi64(uint64_t r0, uint64_t r1, bool &truncated) noexcept {
+  int shl = leading_zeroes(r0);
+  if (shl == 0) {
+    truncated = r1 != 0;
+    return r0;
+  } else {
+    int shr = 64 - shl;
+    truncated = (r1 << shl) != 0;
+    return (r0 << shl) | (r1 >> shr);
+  }
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 uint64_t
+uint32_hi64(uint32_t r0, bool &truncated) noexcept {
+  return uint64_hi64(r0, truncated);
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 uint64_t
+uint32_hi64(uint32_t r0, uint32_t r1, bool &truncated) noexcept {
+  uint64_t x0 = r0;
+  uint64_t x1 = r1;
+  return uint64_hi64((x0 << 32) | x1, truncated);
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 uint64_t
+uint32_hi64(uint32_t r0, uint32_t r1, uint32_t r2, bool &truncated) noexcept {
+  uint64_t x0 = r0;
+  uint64_t x1 = r1;
+  uint64_t x2 = r2;
+  return uint64_hi64(x0, (x1 << 32) | x2, truncated);
+}
+
+// add two small integers, checking for overflow.
+// we want an efficient operation. for msvc, where
+// we don't have built-in intrinsics, this is still
+// pretty fast.
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 limb
+scalar_add(limb x, limb y, bool &overflow) noexcept {
+  limb z;
+// gcc and clang
+#if defined(__has_builtin)
+#if __has_builtin(__builtin_add_overflow)
+  if (!cpp20_and_in_constexpr()) {
+    overflow = __builtin_add_overflow(x, y, &z);
+    return z;
+  }
+#endif
+#endif
+
+  // generic, this still optimizes correctly on MSVC.
+  z = x + y;
+  overflow = z < x;
+  return z;
+}
+
+// multiply two small integers, getting both the high and low bits.
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 limb
+scalar_mul(limb x, limb y, limb &carry) noexcept {
+#ifdef SIMDJSON_FASTFLOAT_64BIT_LIMB
+#if defined(__SIZEOF_INT128__)
+  // GCC and clang both define it as an extension.
+  __uint128_t z = __uint128_t(x) * __uint128_t(y) + __uint128_t(carry);
+  carry = limb(z >> limb_bits);
+  return limb(z);
+#else
+  // fallback, no native 128-bit integer multiplication with carry.
+  // on msvc, this optimizes identically, somehow.
+  value128 z = full_multiplication(x, y);
+  bool overflow;
+  z.low = scalar_add(z.low, carry, overflow);
+  z.high += uint64_t(overflow); // cannot overflow
+  carry = z.high;
+  return z.low;
+#endif
+#else
+  uint64_t z = uint64_t(x) * uint64_t(y) + uint64_t(carry);
+  carry = limb(z >> limb_bits);
+  return limb(z);
+#endif
+}
+
+// add scalar value to bigint starting from offset.
+// used in grade school multiplication
+template <uint16_t size>
+inline SIMDJSON_FASTFLOAT_CONSTEXPR20 bool small_add_from(stackvec<size> &vec, limb y,
+                                                 size_t start) noexcept {
+  size_t index = start;
+  limb carry = y;
+  bool overflow;
+  while (carry != 0 && index < vec.len()) {
+    vec[index] = scalar_add(vec[index], carry, overflow);
+    carry = limb(overflow);
+    index += 1;
+  }
+  if (carry != 0) {
+    SIMDJSON_FASTFLOAT_TRY(vec.try_push(carry));
+  }
   return true;
 }

-// Parses a single digit character and updates the integer value.
-consteval bool parse_digit(const char c, uint64_t &i) {
-  const uint8_t digit = static_cast<uint8_t>(c - '0');
-  if (digit > 9) {
-    return false;
+// add scalar value to bigint.
+template <uint16_t size>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 bool
+small_add(stackvec<size> &vec, limb y) noexcept {
+  return small_add_from(vec, y, 0);
+}
+
+// multiply bigint by scalar value.
+template <uint16_t size>
+inline SIMDJSON_FASTFLOAT_CONSTEXPR20 bool small_mul(stackvec<size> &vec,
+                                            limb y) noexcept {
+  limb carry = 0;
+  for (size_t index = 0; index < vec.len(); index++) {
+    vec[index] = scalar_mul(vec[index], y, carry);
+  }
+  if (carry != 0) {
+    SIMDJSON_FASTFLOAT_TRY(vec.try_push(carry));
   }
-  i = 10 * i + digit;
   return true;
 }

-// Parses a JSON float from a string starting at src.
-// Returns the parsed double and the number of characters consumed.
-consteval std::pair<double, size_t> parse_double(const char *src,
-                                                 const char *end) {
-  auto get_value = [&](const char *pointer) -> char {
-    if (pointer == end) {
-      return '\0';
+// add bigint to bigint starting from index.
+// used in grade school multiplication
+template <uint16_t size>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 bool large_add_from(stackvec<size> &x, limb_span y,
+                                          size_t start) noexcept {
+  // the effective x buffer is from `xstart..x.len()`, so exit early
+  // if we can't get that current range.
+  if (x.len() < start || y.len() > x.len() - start) {
+    SIMDJSON_FASTFLOAT_TRY(x.try_resize(y.len() + start, 0));
+  }
+
+  bool carry = false;
+  for (size_t index = 0; index < y.len(); index++) {
+    limb xi = x[index + start];
+    limb yi = y[index];
+    bool c1 = false;
+    bool c2 = false;
+    xi = scalar_add(xi, yi, c1);
+    if (carry) {
+      xi = scalar_add(xi, 1, c2);
     }
-    return *pointer;
+    x[index + start] = xi;
+    carry = c1 | c2;
+  }
+
+  // handle overflow
+  if (carry) {
+    SIMDJSON_FASTFLOAT_TRY(small_add_from(x, 1, y.len() + start));
+  }
+  return true;
+}
+
+// add bigint to bigint.
+template <uint16_t size>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 bool
+large_add_from(stackvec<size> &x, limb_span y) noexcept {
+  return large_add_from(x, y, 0);
+}
+
+// grade-school multiplication algorithm
+template <uint16_t size>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 bool long_mul(stackvec<size> &x, limb_span y) noexcept {
+  limb_span xs = limb_span(x.data, x.len());
+  stackvec<size> z(xs);
+  limb_span zs = limb_span(z.data, z.len());
+
+  if (y.len() != 0) {
+    limb y0 = y[0];
+    SIMDJSON_FASTFLOAT_TRY(small_mul(x, y0));
+    for (size_t index = 1; index < y.len(); index++) {
+      limb yi = y[index];
+      stackvec<size> zi;
+      if (yi != 0) {
+        // re-use the same buffer throughout
+        zi.set_len(0);
+        SIMDJSON_FASTFLOAT_TRY(zi.try_extend(zs));
+        SIMDJSON_FASTFLOAT_TRY(small_mul(zi, yi));
+        limb_span zis = limb_span(zi.data, zi.len());
+        SIMDJSON_FASTFLOAT_TRY(large_add_from(x, zis, index));
+      }
+    }
+  }
+
+  x.normalize();
+  return true;
+}
+
+// grade-school multiplication algorithm
+template <uint16_t size>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 bool large_mul(stackvec<size> &x, limb_span y) noexcept {
+  if (y.len() == 1) {
+    SIMDJSON_FASTFLOAT_TRY(small_mul(x, y[0]));
+  } else {
+    SIMDJSON_FASTFLOAT_TRY(long_mul(x, y));
+  }
+  return true;
+}
+
+template <typename = void> struct pow5_tables {
+  static constexpr uint32_t large_step = 135;
+  static constexpr uint64_t small_power_of_5[] = {
+      1UL,
+      5UL,
+      25UL,
+      125UL,
+      625UL,
+      3125UL,
+      15625UL,
+      78125UL,
+      390625UL,
+      1953125UL,
+      9765625UL,
+      48828125UL,
+      244140625UL,
+      1220703125UL,
+      6103515625UL,
+      30517578125UL,
+      152587890625UL,
+      762939453125UL,
+      3814697265625UL,
+      19073486328125UL,
+      95367431640625UL,
+      476837158203125UL,
+      2384185791015625UL,
+      11920928955078125UL,
+      59604644775390625UL,
+      298023223876953125UL,
+      1490116119384765625UL,
+      7450580596923828125UL,
   };
-  const char *srcinit = src;
-  bool negative = (get_value(src) == '-');
-  src += uint8_t(negative);
-  uint64_t i = 0;
-  const char *p = src;
-  p += parse_digit(get_value(p), i);
-  bool leading_zero = (i == 0);
-  while (parse_digit(get_value(p), i)) {
-    p++;
+#ifdef SIMDJSON_FASTFLOAT_64BIT_LIMB
+  constexpr static limb large_power_of_5[] = {
+      1414648277510068013UL, 9180637584431281687UL, 4539964771860779200UL,
+      10482974169319127550UL, 198276706040285095UL};
+#else
+  constexpr static limb large_power_of_5[] = {
+      4279965485U, 329373468U,  4020270615U, 2137533757U, 4287402176U,
+      1057042919U, 1071430142U, 2440757623U, 381945767U,  46164893U};
+#endif
+};
+
+#if SIMDJSON_FASTFLOAT_DETAIL_MUST_DEFINE_CONSTEXPR_VARIABLE
+
+template <typename T> constexpr uint32_t pow5_tables<T>::large_step;
+
+template <typename T> constexpr uint64_t pow5_tables<T>::small_power_of_5[];
+
+template <typename T> constexpr limb pow5_tables<T>::large_power_of_5[];
+
+#endif
+
+// big integer type. implements a small subset of big integer
+// arithmetic, using simple algorithms since asymptotically
+// faster algorithms are slower for a small number of limbs.
+// all operations assume the big-integer is normalized.
+struct bigint : pow5_tables<> {
+  // storage of the limbs, in little-endian order.
+  stackvec<bigint_limbs> vec;
+
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 bigint() : vec() {}
+
+  bigint(bigint const &) = delete;
+  bigint &operator=(bigint const &) = delete;
+  bigint(bigint &&) = delete;
+  bigint &operator=(bigint &&other) = delete;
+
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 bigint(uint64_t value) : vec() {
+#ifdef SIMDJSON_FASTFLOAT_64BIT_LIMB
+    vec.push_unchecked(value);
+#else
+    vec.push_unchecked(uint32_t(value));
+    vec.push_unchecked(uint32_t(value >> 32));
+#endif
+    vec.normalize();
   }
-  if (p == src) {
-    simdjson_consteval_error("Invalid float value");
+
+  // get the high 64 bits from the vector, and if bits were truncated.
+  // this is to get the significant digits for the float.
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 uint64_t hi64(bool &truncated) const noexcept {
+#ifdef SIMDJSON_FASTFLOAT_64BIT_LIMB
+    if (vec.len() == 0) {
+      return empty_hi64(truncated);
+    } else if (vec.len() == 1) {
+      return uint64_hi64(vec.rindex(0), truncated);
+    } else {
+      uint64_t result = uint64_hi64(vec.rindex(0), vec.rindex(1), truncated);
+      truncated |= vec.nonzero(2);
+      return result;
+    }
+#else
+    if (vec.len() == 0) {
+      return empty_hi64(truncated);
+    } else if (vec.len() == 1) {
+      return uint32_hi64(vec.rindex(0), truncated);
+    } else if (vec.len() == 2) {
+      return uint32_hi64(vec.rindex(0), vec.rindex(1), truncated);
+    } else {
+      uint64_t result =
+          uint32_hi64(vec.rindex(0), vec.rindex(1), vec.rindex(2), truncated);
+      truncated |= vec.nonzero(3);
+      return result;
+    }
+#endif
   }
-  if ((leading_zero && p != src + 1)) {
-    simdjson_consteval_error("Invalid float value");
+
+  // compare two big integers, returning the large value.
+  // assumes both are normalized. if the return value is
+  // negative, other is larger, if the return value is
+  // positive, this is larger, otherwise they are equal.
+  // the limbs are stored in little-endian order, so we
+  // must compare the limbs in ever order.
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 int compare(bigint const &other) const noexcept {
+    if (vec.len() > other.vec.len()) {
+      return 1;
+    } else if (vec.len() < other.vec.len()) {
+      return -1;
+    } else {
+      for (size_t index = vec.len(); index > 0; index--) {
+        limb xi = vec[index - 1];
+        limb yi = other.vec[index - 1];
+        if (xi > yi) {
+          return 1;
+        } else if (xi < yi) {
+          return -1;
+        }
+      }
+      return 0;
+    }
   }
-  int64_t exponent = 0;
-  bool overflow;
-  if (get_value(p) == '.') {
-    p++;
-    const char *start_decimal_digits = p;
-    if (!parse_digit(get_value(p), i)) {
-      simdjson_consteval_error("Invalid float value");
-    } // no decimal digits
-    p++;
-    while (parse_digit(get_value(p), i)) {
-      p++;
+
+  // shift left each limb n bits, carrying over to the new limb
+  // returns true if we were able to shift all the digits.
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 bool shl_bits(size_t n) noexcept {
+    // Internally, for each item, we shift left by n, and add the previous
+    // right shifted limb-bits.
+    // For example, we transform (for u8) shifted left 2, to:
+    //      b10100100 b01000010
+    //      b10 b10010001 b00001000
+    SIMDJSON_FASTFLOAT_DEBUG_ASSERT(n != 0);
+    SIMDJSON_FASTFLOAT_DEBUG_ASSERT(n < sizeof(limb) * 8);
+
+    size_t shl = n;
+    size_t shr = limb_bits - shl;
+    limb prev = 0;
+    for (size_t index = 0; index < vec.len(); index++) {
+      limb xi = vec[index];
+      vec[index] = (xi << shl) | (prev >> shr);
+      prev = xi;
     }
-    exponent = -(p - start_decimal_digits);
-    overflow = p - src - 1 > 19;
-    if (overflow && leading_zero) {
-      const char *start_digits = src + 2;
-      while (get_value(start_digits) == '0') {
-        start_digits++;
+
+    limb carry = prev >> shr;
+    if (carry != 0) {
+      return vec.try_push(carry);
+    }
+    return true;
+  }
+
+  // move the limbs left by `n` limbs.
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 bool shl_limbs(size_t n) noexcept {
+    SIMDJSON_FASTFLOAT_DEBUG_ASSERT(n != 0);
+    if (n + vec.len() > vec.capacity()) {
+      return false;
+    } else if (!vec.is_empty()) {
+      // move limbs
+      limb *dst = vec.data + n;
+      limb const *src = vec.data;
+      std::copy_backward(src, src + vec.len(), dst + vec.len());
+      // fill in empty limbs
+      limb *first = vec.data;
+      limb *last = first + n;
+      ::std::fill(first, last, 0);
+      vec.set_len(n + vec.len());
+      return true;
+    } else {
+      return true;
+    }
+  }
+
+  // move the limbs left by `n` bits.
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 bool shl(size_t n) noexcept {
+    size_t rem = n % limb_bits;
+    size_t div = n / limb_bits;
+    if (rem != 0) {
+      SIMDJSON_FASTFLOAT_TRY(shl_bits(rem));
+    }
+    if (div != 0) {
+      SIMDJSON_FASTFLOAT_TRY(shl_limbs(div));
+    }
+    return true;
+  }
+
+  // get the number of leading zeros in the bigint.
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 int ctlz() const noexcept {
+    if (vec.is_empty()) {
+      return 0;
+    } else {
+#ifdef SIMDJSON_FASTFLOAT_64BIT_LIMB
+      return leading_zeroes(vec.rindex(0));
+#else
+      // no use defining a specialized leading_zeroes for a 32-bit type.
+      uint64_t r0 = vec.rindex(0);
+      return leading_zeroes(r0 << 32);
+#endif
+    }
+  }
+
+  // get the number of bits in the bigint.
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 int bit_length() const noexcept {
+    int lz = ctlz();
+    return int(limb_bits * vec.len()) - lz;
+  }
+
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 bool mul(limb y) noexcept { return small_mul(vec, y); }
+
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 bool add(limb y) noexcept { return small_add(vec, y); }
+
+  // multiply as if by 2 raised to a power.
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 bool pow2(uint32_t exp) noexcept { return shl(exp); }
+
+  // multiply as if by 5 raised to a power.
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 bool pow5(uint32_t exp) noexcept {
+    // multiply by a power of 5
+    size_t large_length = sizeof(large_power_of_5) / sizeof(limb);
+    limb_span large = limb_span(large_power_of_5, large_length);
+    while (exp >= large_step) {
+      SIMDJSON_FASTFLOAT_TRY(large_mul(vec, large));
+      exp -= large_step;
+    }
+#ifdef SIMDJSON_FASTFLOAT_64BIT_LIMB
+    uint32_t small_step = 27;
+    limb max_native = 7450580596923828125UL;
+#else
+    uint32_t small_step = 13;
+    limb max_native = 1220703125U;
+#endif
+    while (exp >= small_step) {
+      SIMDJSON_FASTFLOAT_TRY(small_mul(vec, max_native));
+      exp -= small_step;
+    }
+    if (exp != 0) {
+      // Work around clang bug https://godbolt.org/z/zedh7rrhc
+      // This is similar to https://github.com/llvm/llvm-project/issues/47746,
+      // except the workaround described there don't work here
+      SIMDJSON_FASTFLOAT_TRY(small_mul(vec, limb((static_cast<void>(small_power_of_5[0]),
+                                         small_power_of_5[exp]))));
+    }
+
+    return true;
+  }
+
+  // multiply as if by 10 raised to a power.
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 bool pow10(uint32_t exp) noexcept {
+    SIMDJSON_FASTFLOAT_TRY(pow5(exp));
+    return pow2(exp);
+  }
+};
+
+} // namespace simdjson_fast_float
+
+#endif
+
+#ifndef SIMDJSON_FASTFLOAT_DIGIT_COMPARISON_H
+#define SIMDJSON_FASTFLOAT_DIGIT_COMPARISON_H
+
+#include <cstdint>
+#include <cstring>
+#include <iterator>
+
+
+namespace simdjson_fast_float {
+
+// 1e0 to 1e19
+constexpr static uint64_t powers_of_ten_uint64[] = {1UL,
+                                                    10UL,
+                                                    100UL,
+                                                    1000UL,
+                                                    10000UL,
+                                                    100000UL,
+                                                    1000000UL,
+                                                    10000000UL,
+                                                    100000000UL,
+                                                    1000000000UL,
+                                                    10000000000UL,
+                                                    100000000000UL,
+                                                    1000000000000UL,
+                                                    10000000000000UL,
+                                                    100000000000000UL,
+                                                    1000000000000000UL,
+                                                    10000000000000000UL,
+                                                    100000000000000000UL,
+                                                    1000000000000000000UL,
+                                                    10000000000000000000UL};
+
+// calculate the exponent, in scientific notation, of the number.
+// this algorithm is not even close to optimized, but it has no practical
+// effect on performance: in order to have a faster algorithm, we'd need
+// to slow down performance for faster algorithms, and this is still fast.
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 int32_t
+scientific_exponent(uint64_t mantissa, int32_t exponent) noexcept {
+  while (mantissa >= 10000) {
+    mantissa /= 10000;
+    exponent += 4;
+  }
+  while (mantissa >= 100) {
+    mantissa /= 100;
+    exponent += 2;
+  }
+  while (mantissa >= 10) {
+    mantissa /= 10;
+    exponent += 1;
+  }
+  return exponent;
+}
+
+// this converts a native floating-point number to an extended-precision float.
+template <typename T>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 adjusted_mantissa
+to_extended(T value) noexcept {
+  using equiv_uint = equiv_uint_t<T>;
+  constexpr equiv_uint exponent_mask = binary_format<T>::exponent_mask();
+  constexpr equiv_uint mantissa_mask = binary_format<T>::mantissa_mask();
+  constexpr equiv_uint hidden_bit_mask = binary_format<T>::hidden_bit_mask();
+
+  adjusted_mantissa am;
+  int32_t bias = binary_format<T>::mantissa_explicit_bits() -
+                 binary_format<T>::minimum_exponent();
+  equiv_uint bits;
+#if SIMDJSON_FASTFLOAT_HAS_BIT_CAST
+  bits = std::bit_cast<equiv_uint>(value);
+#else
+  ::memcpy(&bits, &value, sizeof(T));
+#endif
+  if ((bits & exponent_mask) == 0) {
+    // denormal
+    am.power2 = 1 - bias;
+    am.mantissa = bits & mantissa_mask;
+  } else {
+    // normal
+    am.power2 = int32_t((bits & exponent_mask) >>
+                        binary_format<T>::mantissa_explicit_bits());
+    am.power2 -= bias;
+    am.mantissa = (bits & mantissa_mask) | hidden_bit_mask;
+  }
+
+  return am;
+}
+
+// get the extended precision value of the halfway point between b and b+u.
+// we are given a native float that represents b, so we need to adjust it
+// halfway between b and b+u.
+template <typename T>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 adjusted_mantissa
+to_extended_halfway(T value) noexcept {
+  adjusted_mantissa am = to_extended(value);
+  am.mantissa <<= 1;
+  am.mantissa += 1;
+  am.power2 -= 1;
+  return am;
+}
+
+// round an extended-precision float to the nearest machine float.
+template <typename T, typename callback>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 void round(adjusted_mantissa &am,
+                                                         callback cb) noexcept {
+  int32_t mantissa_shift = 64 - binary_format<T>::mantissa_explicit_bits() - 1;
+  if (-am.power2 >= mantissa_shift) {
+    // have a denormal float
+    int32_t shift = -am.power2 + 1;
+    cb(am, (shift < 64 ? shift : 64));
+    // check for round-up: if rounding-nearest carried us to the hidden bit.
+    am.power2 = (am.mantissa <
+                 (uint64_t(1) << binary_format<T>::mantissa_explicit_bits()))
+                    ? 0
+                    : 1;
+    return;
+  }
+
+  // have a normal float, use the default shift.
+  cb(am, mantissa_shift);
+
+  // check for carry
+  if (am.mantissa >=
+      (uint64_t(2) << binary_format<T>::mantissa_explicit_bits())) {
+    am.mantissa = (uint64_t(1) << binary_format<T>::mantissa_explicit_bits());
+    am.power2++;
+  }
+
+  // check for infinite: we could have carried to an infinite power
+  am.mantissa &= ~(uint64_t(1) << binary_format<T>::mantissa_explicit_bits());
+  if (am.power2 >= binary_format<T>::infinite_power()) {
+    am.power2 = binary_format<T>::infinite_power();
+    am.mantissa = 0;
+  }
+}
+
+template <typename callback>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 void
+round_nearest_tie_even(adjusted_mantissa &am, int32_t shift,
+                       callback cb) noexcept {
+  uint64_t const mask = (shift == 64) ? UINT64_MAX : (uint64_t(1) << shift) - 1;
+  uint64_t const halfway = (shift == 0) ? 0 : uint64_t(1) << (shift - 1);
+  uint64_t truncated_bits = am.mantissa & mask;
+  bool is_above = truncated_bits > halfway;
+  bool is_halfway = truncated_bits == halfway;
+
+  // shift digits into position
+  if (shift == 64) {
+    am.mantissa = 0;
+  } else {
+    am.mantissa >>= shift;
+  }
+  am.power2 += shift;
+
+  bool is_odd = (am.mantissa & 1) == 1;
+  am.mantissa += uint64_t(cb(is_odd, is_halfway, is_above));
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 void
+round_down(adjusted_mantissa &am, int32_t shift) noexcept {
+  if (shift == 64) {
+    am.mantissa = 0;
+  } else {
+    am.mantissa >>= shift;
+  }
+  am.power2 += shift;
+}
+
+template <typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 void
+skip_zeros(UC const *&first, UC const *last) noexcept {
+  uint64_t val;
+  while (!cpp20_and_in_constexpr() &&
+         std::distance(first, last) >= int_cmp_len<UC>()) {
+    ::memcpy(&val, first, sizeof(uint64_t));
+    if (val != int_cmp_zeros<UC>()) {
+      break;
+    }
+    first += int_cmp_len<UC>();
+  }
+  while (first != last) {
+    if (*first != UC('0')) {
+      break;
+    }
+    first++;
+  }
+}
+
+// determine if any non-zero digits were truncated.
+// all characters must be valid digits.
+template <typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 bool
+is_truncated(UC const *first, UC const *last) noexcept {
+  // do 8-bit optimizations, can just compare to 8 literal 0s.
+  uint64_t val;
+  while (!cpp20_and_in_constexpr() &&
+         std::distance(first, last) >= int_cmp_len<UC>()) {
+    ::memcpy(&val, first, sizeof(uint64_t));
+    if (val != int_cmp_zeros<UC>()) {
+      return true;
+    }
+    first += int_cmp_len<UC>();
+  }
+  while (first != last) {
+    if (*first != UC('0')) {
+      return true;
+    }
+    ++first;
+  }
+  return false;
+}
+
+template <typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 bool
+is_truncated(span<UC const> s) noexcept {
+  return is_truncated(s.ptr, s.ptr + s.len());
+}
+
+template <typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 void
+parse_eight_digits(UC const *&p, limb &value, size_t &counter,
+                   size_t &count) noexcept {
+  value = value * 100000000 + parse_eight_digits_unrolled(p);
+  p += 8;
+  counter += 8;
+  count += 8;
+}
+
+template <typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR14 void
+parse_one_digit(UC const *&p, limb &value, size_t &counter,
+                size_t &count) noexcept {
+  value = value * 10 + limb(*p - UC('0'));
+  p++;
+  counter++;
+  count++;
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 void
+add_native(bigint &big, limb power, limb value) noexcept {
+  big.mul(power);
+  big.add(value);
+}
+
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 void
+round_up_bigint(bigint &big, size_t &count) noexcept {
+  // need to round-up the digits, but need to avoid rounding
+  // ....9999 to ...10000, which could cause a false halfway point.
+  add_native(big, 10, 1);
+  count++;
+}
+
+// parse the significant digits into a big integer
+template <typename UC>
+inline SIMDJSON_FASTFLOAT_CONSTEXPR20 void
+parse_mantissa(bigint &result, parsed_number_string_t<UC> &num,
+               size_t max_digits, size_t &digits) noexcept {
+  // try to minimize the number of big integer and scalar multiplication.
+  // therefore, try to parse 8 digits at a time, and multiply by the largest
+  // scalar value (9 or 19 digits) for each step.
+  size_t counter = 0;
+  digits = 0;
+  limb value = 0;
+#ifdef SIMDJSON_FASTFLOAT_64BIT_LIMB
+  size_t step = 19;
+#else
+  size_t step = 9;
+#endif
+
+  // process all integer digits.
+  UC const *p = num.integer.ptr;
+  UC const *pend = p + num.integer.len();
+  skip_zeros(p, pend);
+  // process all digits, in increments of step per loop
+  while (p != pend) {
+    while ((std::distance(p, pend) >= 8) && (step - counter >= 8) &&
+           (max_digits - digits >= 8)) {
+      parse_eight_digits(p, value, counter, digits);
+    }
+    while (counter < step && p != pend && digits < max_digits) {
+      parse_one_digit(p, value, counter, digits);
+    }
+    if (digits == max_digits) {
+      // add the temporary value, then check if we've truncated any digits
+      add_native(result, limb(powers_of_ten_uint64[counter]), value);
+      bool truncated = is_truncated(p, pend);
+      if (num.fraction.ptr != nullptr) {
+        truncated |= is_truncated(num.fraction);
+      }
+      if (truncated) {
+        round_up_bigint(result, digits);
+      }
+      return;
+    } else {
+      add_native(result, limb(powers_of_ten_uint64[counter]), value);
+      counter = 0;
+      value = 0;
+    }
+  }
+
+  // add our fraction digits, if they're available.
+  if (num.fraction.ptr != nullptr) {
+    p = num.fraction.ptr;
+    pend = p + num.fraction.len();
+    if (digits == 0) {
+      skip_zeros(p, pend);
+    }
+    // process all digits, in increments of step per loop
+    while (p != pend) {
+      while ((std::distance(p, pend) >= 8) && (step - counter >= 8) &&
+             (max_digits - digits >= 8)) {
+        parse_eight_digits(p, value, counter, digits);
+      }
+      while (counter < step && p != pend && digits < max_digits) {
+        parse_one_digit(p, value, counter, digits);
+      }
+      if (digits == max_digits) {
+        // add the temporary value, then check if we've truncated any digits
+        add_native(result, limb(powers_of_ten_uint64[counter]), value);
+        bool truncated = is_truncated(p, pend);
+        if (truncated) {
+          round_up_bigint(result, digits);
+        }
+        return;
+      } else {
+        add_native(result, limb(powers_of_ten_uint64[counter]), value);
+        counter = 0;
+        value = 0;
       }
-      overflow = p - start_digits > 19;
     }
+  }
+
+  if (counter != 0) {
+    add_native(result, limb(powers_of_ten_uint64[counter]), value);
+  }
+}
+
+template <typename T>
+inline SIMDJSON_FASTFLOAT_CONSTEXPR20 adjusted_mantissa
+positive_digit_comp(bigint &bigmant, int32_t exponent) noexcept {
+  SIMDJSON_FASTFLOAT_ASSERT(bigmant.pow10(uint32_t(exponent)));
+  adjusted_mantissa answer;
+  bool truncated;
+  answer.mantissa = bigmant.hi64(truncated);
+  int bias = binary_format<T>::mantissa_explicit_bits() -
+             binary_format<T>::minimum_exponent();
+  answer.power2 = bigmant.bit_length() - 64 + bias;
+
+  round<T>(answer, [truncated](adjusted_mantissa &a, int32_t shift) {
+    round_nearest_tie_even(
+        a, shift,
+        [truncated](bool is_odd, bool is_halfway, bool is_above) -> bool {
+          return is_above || (is_halfway && truncated) ||
+                 (is_odd && is_halfway);
+        });
+  });
+
+  return answer;
+}
+
+// the scaling here is quite simple: we have, for the real digits `m * 10^e`,
+// and for the theoretical digits `n * 2^f`. Since `e` is always negative,
+// to scale them identically, we do `n * 2^f * 5^-f`, so we now have `m * 2^e`.
+// we then need to scale by `2^(f- e)`, and then the two significant digits
+// are of the same magnitude.
+template <typename T>
+inline SIMDJSON_FASTFLOAT_CONSTEXPR20 adjusted_mantissa negative_digit_comp(
+    bigint &bigmant, adjusted_mantissa am, int32_t exponent) noexcept {
+  bigint &real_digits = bigmant;
+  int32_t real_exp = exponent;
+
+  // get the value of `b`, rounded down, and get a bigint representation of b+h
+  adjusted_mantissa am_b = am;
+  // gcc7 buf: use a lambda to remove the noexcept qualifier bug with
+  // -Wnoexcept-type.
+  round<T>(am_b,
+           [](adjusted_mantissa &a, int32_t shift) { round_down(a, shift); });
+  T b;
+  to_float(false, am_b, b);
+  adjusted_mantissa theor = to_extended_halfway(b);
+  bigint theor_digits(theor.mantissa);
+  int32_t theor_exp = theor.power2;
+
+  // scale real digits and theor digits to be same power.
+  int32_t pow2_exp = theor_exp - real_exp;
+  uint32_t pow5_exp = uint32_t(-real_exp);
+  if (pow5_exp != 0) {
+    SIMDJSON_FASTFLOAT_ASSERT(theor_digits.pow5(pow5_exp));
+  }
+  if (pow2_exp > 0) {
+    SIMDJSON_FASTFLOAT_ASSERT(theor_digits.pow2(uint32_t(pow2_exp)));
+  } else if (pow2_exp < 0) {
+    SIMDJSON_FASTFLOAT_ASSERT(real_digits.pow2(uint32_t(-pow2_exp)));
+  }
+
+  // compare digits, and use it to direct rounding
+  int ord = real_digits.compare(theor_digits);
+  adjusted_mantissa answer = am;
+  round<T>(answer, [ord](adjusted_mantissa &a, int32_t shift) {
+    round_nearest_tie_even(
+        a, shift, [ord](bool is_odd, bool _, bool __) -> bool {
+          static_cast<void>(_);  // not needed, since we've done our comparison
+          static_cast<void>(__); // not needed, since we've done our comparison
+          if (ord > 0) {
+            return true;
+          } else if (ord < 0) {
+            return false;
+          } else {
+            return is_odd;
+          }
+        });
+  });
+
+  return answer;
+}
+
+// parse the significant digits as a big integer to unambiguously round
+// the significant digits. here, we are trying to determine how to round
+// an extended float representation close to `b+h`, halfway between `b`
+// (the float rounded-down) and `b+u`, the next positive float. this
+// algorithm is always correct, and uses one of two approaches. when
+// the exponent is positive relative to the significant digits (such as
+// 1234), we create a big-integer representation, get the high 64-bits,
+// determine if any lower bits are truncated, and use that to direct
+// rounding. in case of a negative exponent relative to the significant
+// digits (such as 1.2345), we create a theoretical representation of
+// `b` as a big-integer type, scaled to the same binary exponent as
+// the actual digits. we then compare the big integer representations
+// of both, and use that to direct rounding.
+template <typename T, typename UC>
+inline SIMDJSON_FASTFLOAT_CONSTEXPR20 adjusted_mantissa
+digit_comp(parsed_number_string_t<UC> &num, adjusted_mantissa am) noexcept {
+  // remove the invalid exponent bias
+  am.power2 -= invalid_am_bias;
+
+  int32_t sci_exp =
+      scientific_exponent(num.mantissa, static_cast<int32_t>(num.exponent));
+  size_t max_digits = binary_format<T>::max_digits();
+  size_t digits = 0;
+  bigint bigmant;
+  parse_mantissa(bigmant, num, max_digits, digits);
+  // can't underflow, since digits is at most max_digits.
+  int32_t exponent = sci_exp + 1 - int32_t(digits);
+  if (exponent >= 0) {
+    return positive_digit_comp<T>(bigmant, exponent);
   } else {
-    overflow = p - src > 19;
+    return negative_digit_comp<T>(bigmant, am, exponent);
+  }
+}
+
+} // namespace simdjson_fast_float
+
+#endif
+
+#ifndef SIMDJSON_FASTFLOAT_PARSE_NUMBER_H
+#define SIMDJSON_FASTFLOAT_PARSE_NUMBER_H
+
+
+#include <cmath>
+#include <cstring>
+#include <limits>
+#include <system_error>
+
+namespace simdjson_fast_float {
+
+namespace detail {
+/**
+ * Special case +inf, -inf, nan, infinity, -infinity.
+ * The case comparisons could be made much faster given that we know that the
+ * strings a null-free and fixed.
+ **/
+template <typename T, typename UC>
+from_chars_result_t<UC>
+    SIMDJSON_FASTFLOAT_CONSTEXPR14 parse_infnan(UC const *first, UC const *last,
+                                       T &value, chars_format fmt) noexcept {
+  from_chars_result_t<UC> answer{};
+  answer.ptr = first;
+  answer.ec = std::errc(); // be optimistic
+  // assume first < last, so dereference without checks;
+  bool const minusSign = (*first == UC('-'));
+  // C++17 20.19.3.(7.1) explicitly forbids '+' sign here
+  if ((*first == UC('-')) ||
+      (uint64_t(fmt & chars_format::allow_leading_plus) &&
+       (*first == UC('+')))) {
+    ++first;
+  }
+  if (last - first >= 3) {
+    if (simdjson_fastfloat_strncasecmp3(first, str_const_nan<UC>())) {
+      answer.ptr = (first += 3);
+      value = minusSign ? -std::numeric_limits<T>::quiet_NaN()
+                        : std::numeric_limits<T>::quiet_NaN();
+      // Check for possible nan(n-char-seq-opt), C++17 20.19.3.7,
+      // C11 7.20.1.3.3. At least MSVC produces nan(ind) and nan(snan).
+      if (first != last && *first == UC('(')) {
+        for (UC const *ptr = first + 1; ptr != last; ++ptr) {
+          if (*ptr == UC(')')) {
+            answer.ptr = ptr + 1; // valid nan(n-char-seq-opt)
+            break;
+          } else if (!((UC('a') <= *ptr && *ptr <= UC('z')) ||
+                       (UC('A') <= *ptr && *ptr <= UC('Z')) ||
+                       (UC('0') <= *ptr && *ptr <= UC('9')) || *ptr == UC('_')))
+            break; // forbidden char, not nan(n-char-seq-opt)
+        }
+      }
+      return answer;
+    }
+    if (simdjson_fastfloat_strncasecmp3(first, str_const_inf<UC>())) {
+      if ((last - first >= 8) &&
+          simdjson_fastfloat_strncasecmp5(first + 3, str_const_inf<UC>() + 3)) {
+        answer.ptr = first + 8;
+      } else {
+        answer.ptr = first + 3;
+      }
+      value = minusSign ? -std::numeric_limits<T>::infinity()
+                        : std::numeric_limits<T>::infinity();
+      return answer;
+    }
   }
-  if (overflow) {
-    simdjson_consteval_error(
-        "Overflow while computing the float value: too many digits");
+  answer.ec = std::errc::invalid_argument;
+  return answer;
+}
+
+/**
+ * Returns true if the floating-pointing rounding mode is to 'nearest'.
+ * It is the default on most system. This function is meant to be inexpensive.
+ * Credit : @mwalcott3
+ */
+simdjson_fastfloat_really_inline bool rounds_to_nearest() noexcept {
+  // https://lemire.me/blog/2020/06/26/gcc-not-nearest/
+#if (FLT_EVAL_METHOD != 1) && (FLT_EVAL_METHOD != 0)
+  return false;
+#endif
+  // See
+  // A fast function to check your floating-point rounding mode
+  // https://lemire.me/blog/2022/11/16/a-fast-function-to-check-your-floating-point-rounding-mode/
+  //
+  // This function is meant to be equivalent to :
+  // prior: #include <cfenv>
+  //  return fegetround() == FE_TONEAREST;
+  // However, it is expected to be much faster than the fegetround()
+  // function call.
+  //
+  // The volatile keyword prevents the compiler from computing the function
+  // at compile-time.
+  // There might be other ways to prevent compile-time optimizations (e.g.,
+  // asm). The value does not need to be std::numeric_limits<float>::min(), any
+  // small value so that 1 + x should round to 1 would do (after accounting for
+  // excess precision, as in 387 instructions).
+  static float volatile fmin = (std::numeric_limits<float>::min)();
+  float fmini = fmin; // we copy it so that it gets loaded at most once.
+//
+// Explanation:
+// Only when fegetround() == FE_TONEAREST do we have that
+// fmin + 1.0f == 1.0f - fmin.
+//
+// FE_UPWARD:
+//  fmin + 1.0f > 1
+//  1.0f - fmin == 1
+//
+// FE_DOWNWARD or  FE_TOWARDZERO:
+//  fmin + 1.0f == 1
+//  1.0f - fmin < 1
+//
+// Note: This may fail to be accurate if fast-math has been
+// enabled, as rounding conventions may not apply.
+#ifdef SIMDJSON_FASTFLOAT_VISUAL_STUDIO
+#pragma warning(push)
+//  todo: is there a VS warning?
+//  see
+//  https://stackoverflow.com/questions/46079446/is-there-a-warning-for-floating-point-equality-checking-in-visual-studio-2013
+#elif defined(__clang__)
+#pragma clang diagnostic push
+#pragma clang diagnostic ignored "-Wfloat-equal"
+#elif defined(__GNUC__)
+#pragma GCC diagnostic push
+#pragma GCC diagnostic ignored "-Wfloat-equal"
+#endif
+  return (fmini + 1.0f == 1.0f - fmini);
+#ifdef SIMDJSON_FASTFLOAT_VISUAL_STUDIO
+#pragma warning(pop)
+#elif defined(__clang__)
+#pragma clang diagnostic pop
+#elif defined(__GNUC__)
+#pragma GCC diagnostic pop
+#endif
+}
+
+} // namespace detail
+
+template <typename T> struct from_chars_caller {
+  template <typename UC>
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 static from_chars_result_t<UC>
+  call(UC const *first, UC const *last, T &value,
+       parse_options_t<UC> options) noexcept {
+    return from_chars_advanced(first, last, value, options);
   }
-  if (get_value(p) == 'e' || get_value(p) == 'E') {
-    p++;
-    bool exp_neg = get_value(p) == '-';
-    p += exp_neg || get_value(p) == '+';
-    uint64_t exp = 0;
-    const char *start_exp_digits = p;
-    while (parse_digit(get_value(p), exp)) {
-      p++;
+};
+
+#ifdef __STDCPP_FLOAT32_T__
+template <> struct from_chars_caller<std::float32_t> {
+  template <typename UC>
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 static from_chars_result_t<UC>
+  call(UC const *first, UC const *last, std::float32_t &value,
+       parse_options_t<UC> options) noexcept {
+    // if std::float32_t is defined, and we are in C++23 mode; macro set for
+    // float32; set value to float due to equivalence between float and
+    // float32_t
+    float val = 0.0f;
+    auto ret = from_chars_advanced(first, last, val, options);
+    value = val;
+    return ret;
+  }
+};
+#endif
+
+#ifdef __STDCPP_FLOAT64_T__
+template <> struct from_chars_caller<std::float64_t> {
+  template <typename UC>
+  SIMDJSON_FASTFLOAT_CONSTEXPR20 static from_chars_result_t<UC>
+  call(UC const *first, UC const *last, std::float64_t &value,
+       parse_options_t<UC> options) noexcept {
+    // if std::float64_t is defined, and we are in C++23 mode; macro set for
+    // float64; set value as double due to equivalence between double and
+    // float64_t
+    double val = 0.0;
+    auto ret = from_chars_advanced(first, last, val, options);
+    value = val;
+    return ret;
+  }
+};
+#endif
+
+template <typename T, typename UC, typename>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+from_chars(UC const *first, UC const *last, T &value,
+           chars_format fmt /*= chars_format::general*/) noexcept {
+  return from_chars_caller<T>::call(first, last, value,
+                                    parse_options_t<UC>(fmt));
+}
+
+template <typename T>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 bool
+clinger_fast_path_impl(uint64_t mantissa, int64_t exponent, bool is_negative,
+                       T &value) noexcept {
+  // The implementation of the Clinger's fast path is convoluted because
+  // we want round-to-nearest in all cases, irrespective of the rounding mode
+  // selected on the thread.
+  // We proceed optimistically, assuming that detail::rounds_to_nearest()
+  // returns true.
+  if (binary_format<T>::min_exponent_fast_path() <= exponent &&
+      exponent <= binary_format<T>::max_exponent_fast_path() &&
+      mantissa <= binary_format<T>::max_mantissa_fast_path()) {
+    // The mantissa bound above is a necessary condition for BOTH branches
+    // below: the rounding-mode-dependent branch checks the tighter
+    // max_mantissa_fast_path(exponent) <= max_mantissa_fast_path(). Testing
+    // it before detail::rounds_to_nearest() spares long-mantissa inputs
+    // (which can never take the fast path) the volatile-float probe.
+    //
+    // Unfortunately, the conventional Clinger's fast path is only possible
+    // when the system rounds to the nearest float.
+    //
+    // We expect the next branch to almost always be selected.
+    // We could check it first (before the previous branch), but
+    // there might be performance advantages at having the check
+    // be last.
+    if (!cpp20_and_in_constexpr() && detail::rounds_to_nearest()) {
+      // We have that fegetround() == FE_TONEAREST.
+      // Next is Clinger's fast path.
+      value = T(mantissa);
+      if (exponent < 0) {
+        value = value / binary_format<T>::exact_power_of_ten(-exponent);
+      } else {
+        value = value * binary_format<T>::exact_power_of_ten(exponent);
+      }
+      if (is_negative) {
+        value = -value;
+      }
+      return true;
+    } else {
+      // We do not have that fegetround() == FE_TONEAREST.
+      // Next is a modified Clinger's fast path, inspired by Jakub Jelinek's
+      // proposal
+      if (exponent >= 0 &&
+          mantissa <= binary_format<T>::max_mantissa_fast_path(exponent)) {
+#if defined(__clang__) || defined(SIMDJSON_FASTFLOAT_32BIT)
+        // Clang may map 0 to -0.0 when fegetround() == FE_DOWNWARD
+        if (mantissa == 0) {
+          value = is_negative ? T(-0.) : T(0.);
+          return true;
+        }
+#endif
+        value = T(mantissa) * binary_format<T>::exact_power_of_ten(exponent);
+        if (is_negative) {
+          value = -value;
+        }
+        return true;
+      }
     }
-    if (p - start_exp_digits == 0 || p - start_exp_digits > 19) {
-      simdjson_consteval_error("Invalid float value");
+  }
+  return false;
+}
+
+/**
+ * This function overload takes parsed_number_string_t structure that is created
+ * and populated either by from_chars_advanced function taking chars range and
+ * parsing options or other parsing custom function implemented by user.
+ */
+template <typename T, typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+from_chars_advanced(parsed_number_string_t<UC> &pns, T &value) noexcept {
+  static_assert(is_supported_float_type<T>::value,
+                "only some floating-point types are supported");
+  static_assert(is_supported_char_type<UC>::value,
+                "only char, wchar_t, char16_t and char32_t are supported");
+
+  from_chars_result_t<UC> answer;
+
+  answer.ec = std::errc(); // be optimistic
+  answer.ptr = pns.lastmatch;
+
+  if (!pns.too_many_digits &&
+      clinger_fast_path_impl(pns.mantissa, pns.exponent, pns.negative, value))
+    return answer;
+
+  adjusted_mantissa am =
+      compute_float<binary_format<T>>(pns.exponent, pns.mantissa);
+  if (pns.too_many_digits && am.power2 >= 0) {
+    if (am != compute_float<binary_format<T>>(pns.exponent, pns.mantissa + 1)) {
+      am = compute_error<binary_format<T>>(pns.exponent, pns.mantissa);
     }
-    exponent += exp_neg ? 0 - exp : exp;
   }
+  // If we called compute_float<binary_format<T>>(pns.exponent, pns.mantissa)
+  // and we have an invalid power (am.power2 < 0), then we need to go the long
+  // way around again. This is very uncommon.
+  if (am.power2 < 0) {
+    am = digit_comp<T>(pns, am);
+  }
+  to_float(pns.negative, am, value);
+  // Test for over/underflow.
+  if ((pns.mantissa != 0 && am.mantissa == 0 && am.power2 == 0) ||
+      am.power2 == binary_format<T>::infinite_power()) {
+    answer.ec = std::errc::result_out_of_range;
+  }
+  return answer;
+}

-  overflow = overflow || exponent < simdjson::internal::smallest_power ||
-             exponent > simdjson::internal::largest_power;
-  if (overflow) {
-    simdjson_consteval_error("Overflow while computing the float value");
+// Slow path: re-parse materializing the integer/fraction spans the hot no-span
+// parse skipped, then run the full algorithm. The two callers reach it only
+// through a simdjson_fastfloat_unlikely branch, so the optimizer keeps this re-parse off
+// the hot path on its own (no function-level noinline needed).
+// from_chars_advanced already handles both the too_many_digits disambiguation
+// and the am.power2<0 digit_comp recompute, so both slow branches collapse to
+// one helper call.
+template <typename T, typename UC>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+parse_number_slow_path(UC const *first, UC const *last, T &value,
+                       parse_options_t<UC> options, bool bjf) noexcept {
+  parsed_number_string_t<UC> pns =
+      bjf ? parse_number_string<true, UC>(first, last, options, true)
+          : parse_number_string<false, UC>(first, last, options, true);
+  return from_chars_advanced(pns, value);
+}
+
+template <typename T, typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+from_chars_float_advanced(UC const *first, UC const *last, T &value,
+                          parse_options_t<UC> options) noexcept {
+
+  static_assert(is_supported_float_type<T>::value,
+                "only some floating-point types are supported");
+  static_assert(is_supported_char_type<UC>::value,
+                "only char, wchar_t, char16_t and char32_t are supported");
+
+  chars_format const fmt = detail::adjust_for_feature_macros(options.format);
+
+  from_chars_result_t<UC> answer;
+  if (uint64_t(fmt & chars_format::skip_white_space)) {
+    while ((first != last) && simdjson_fast_float::is_space(*first)) {
+      first++;
+    }
+  }
+  if (first == last) {
+    answer.ec = std::errc::invalid_argument;
+    answer.ptr = first;
+    return answer;
+  }
+  bool const bjf = uint64_t(fmt & detail::basic_json_fmt) != 0;
+
+  // Fast path: parse WITHOUT materializing the integer/fraction spans (read
+  // only by the rare slow paths). Skipping their stores keeps the fat
+  // parsed_number_string_t off the hot path. store_spans is a runtime argument,
+  // so this reuses the single parse_number_string instantiation.
+  parsed_number_string_t<UC> pns =
+      bjf ? parse_number_string<true, UC>(first, last, options, false)
+          : parse_number_string<false, UC>(first, last, options, false);
+  if (!pns.valid) {
+    if (uint64_t(fmt & chars_format::no_infnan)) {
+      answer.ec = std::errc::invalid_argument;
+      answer.ptr = first;
+      return answer;
+    } else {
+      return detail::parse_infnan(first, last, value, fmt);
+    }
   }
-  double d;
-  if (!compute_float_64(exponent, i, negative, d)) {
+
+  // Slow path A (rare): > 19 significant digits. The no-span parse left the
+  // mantissa un-truncated and skipped the span-based recompute; the cold helper
+  // re-parses with spans and runs the full algorithm.
+  //
+// We have to disable -Wc++20-extensions for the [[unlikely]] attribute
+// See comment for @jwakely at
+// https://github.com/fastfloat/simdjson_fast_float/pull/387#discussion_r3366943539
+// This is unfortunate.
+#ifdef __clang__
+#pragma clang diagnostic push
+#if (!defined(__APPLE_CC__) && __clang_major__ >= 10) || (__clang_major__ >= 13)
+#pragma clang diagnostic ignored "-Wc++20-extensions"
+#endif
+#endif
+  if simdjson_fastfloat_unlikely (pns.too_many_digits) {
+    return parse_number_slow_path<T, UC>(first, last, value, options, bjf);
+  }
+  answer.ec = std::errc(); // be optimistic
+  answer.ptr = pns.lastmatch;
+
+  if (clinger_fast_path_impl(pns.mantissa, pns.exponent, pns.negative, value)) {
+    return answer;
+  }
+
+  adjusted_mantissa am =
+      compute_float<binary_format<T>>(pns.exponent, pns.mantissa);
+  // Slow path B (rare): Eisel-Lemire could not resolve; digit_comp needs the
+  // integer/fraction spans. Route to the cold helper (clinger there is a
+  // dead-effect since it already failed here; the cold re-parse + digit_comp
+  // via from_chars_advanced reproduces this branch).
+  if simdjson_fastfloat_unlikely (am.power2 < 0) {
+    return parse_number_slow_path<T, UC>(first, last, value, options, bjf);
+  }
+#ifdef __clang__
+#pragma clang diagnostic pop
+#endif
+  to_float(pns.negative, am, value);
+  // Test for over/underflow.
+  if ((pns.mantissa != 0 && am.mantissa == 0 && am.power2 == 0) ||
+      am.power2 == binary_format<T>::infinite_power()) {
+    answer.ec = std::errc::result_out_of_range;
+  }
+  return answer;
+}
+
+template <typename T, typename UC, typename>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+from_chars(UC const *first, UC const *last, T &value, int base) noexcept {
+
+  static_assert(is_supported_integer_type<T>::value,
+                "only integer types are supported");
+  static_assert(is_supported_char_type<UC>::value,
+                "only char, wchar_t, char16_t and char32_t are supported");
+
+  parse_options_t<UC> options;
+  options.base = base;
+  return from_chars_advanced(first, last, value, options);
+}
+
+template <typename T>
+SIMDJSON_FASTFLOAT_CONSTEXPR20
+    typename std::enable_if<is_supported_float_type<T>::value, T>::type
+    integer_times_pow10(uint64_t mantissa, int decimal_exponent) noexcept {
+  T value;
+  if (clinger_fast_path_impl(mantissa, decimal_exponent, false, value))
+    return value;
+
+  adjusted_mantissa am =
+      compute_float<binary_format<T>>(decimal_exponent, mantissa);
+  to_float(false, am, value);
+  return value;
+}
+
+template <typename T>
+SIMDJSON_FASTFLOAT_CONSTEXPR20
+    typename std::enable_if<is_supported_float_type<T>::value, T>::type
+    integer_times_pow10(int64_t mantissa, int decimal_exponent) noexcept {
+  const bool is_negative = mantissa < 0;
+  const uint64_t m = static_cast<uint64_t>(is_negative ? -mantissa : mantissa);
+
+  T value;
+  if (clinger_fast_path_impl(m, decimal_exponent, is_negative, value))
+    return value;
+
+  adjusted_mantissa am = compute_float<binary_format<T>>(decimal_exponent, m);
+  to_float(is_negative, am, value);
+  return value;
+}
+
+SIMDJSON_FASTFLOAT_CONSTEXPR20 inline double
+integer_times_pow10(uint64_t mantissa, int decimal_exponent) noexcept {
+  return integer_times_pow10<double>(mantissa, decimal_exponent);
+}
+
+SIMDJSON_FASTFLOAT_CONSTEXPR20 inline double
+integer_times_pow10(int64_t mantissa, int decimal_exponent) noexcept {
+  return integer_times_pow10<double>(mantissa, decimal_exponent);
+}
+
+// the following overloads are here to avoid surprising ambiguity for int,
+// unsigned, etc.
+template <typename T, typename Int>
+SIMDJSON_FASTFLOAT_CONSTEXPR20
+    typename std::enable_if<is_supported_float_type<T>::value &&
+                                std::is_integral<Int>::value &&
+                                !std::is_signed<Int>::value,
+                            T>::type
+    integer_times_pow10(Int mantissa, int decimal_exponent) noexcept {
+  return integer_times_pow10<T>(static_cast<uint64_t>(mantissa),
+                                decimal_exponent);
+}
+
+template <typename T, typename Int>
+SIMDJSON_FASTFLOAT_CONSTEXPR20
+    typename std::enable_if<is_supported_float_type<T>::value &&
+                                std::is_integral<Int>::value &&
+                                std::is_signed<Int>::value,
+                            T>::type
+    integer_times_pow10(Int mantissa, int decimal_exponent) noexcept {
+  return integer_times_pow10<T>(static_cast<int64_t>(mantissa),
+                                decimal_exponent);
+}
+
+template <typename Int>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 typename std::enable_if<
+    std::is_integral<Int>::value && !std::is_signed<Int>::value, double>::type
+integer_times_pow10(Int mantissa, int decimal_exponent) noexcept {
+  return integer_times_pow10(static_cast<uint64_t>(mantissa), decimal_exponent);
+}
+
+template <typename Int>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 typename std::enable_if<
+    std::is_integral<Int>::value && std::is_signed<Int>::value, double>::type
+integer_times_pow10(Int mantissa, int decimal_exponent) noexcept {
+  return integer_times_pow10(static_cast<int64_t>(mantissa), decimal_exponent);
+}
+
+template <typename T, typename UC>
+SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+from_chars_int_advanced(UC const *first, UC const *last, T &value,
+                        parse_options_t<UC> options) noexcept {
+
+  static_assert(is_supported_integer_type<T>::value,
+                "only integer types are supported");
+  static_assert(is_supported_char_type<UC>::value,
+                "only char, wchar_t, char16_t and char32_t are supported");
+
+  chars_format const fmt = detail::adjust_for_feature_macros(options.format);
+  int const base = options.base;
+
+  from_chars_result_t<UC> answer;
+  if (uint64_t(fmt & chars_format::skip_white_space)) {
+    while ((first != last) && simdjson_fast_float::is_space(*first)) {
+      first++;
+    }
+  }
+  if (first == last || base < 2 || base > 36) {
+    answer.ec = std::errc::invalid_argument;
+    answer.ptr = first;
+    return answer;
+  }
+
+  return parse_int_string(first, last, value, options);
+}
+
+template <size_t TypeIx> struct from_chars_advanced_caller {
+  static_assert(TypeIx > 0, "unsupported type");
+};
+
+template <> struct from_chars_advanced_caller<1> {
+  template <typename T, typename UC>
+  simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 static from_chars_result_t<UC>
+  call(UC const *first, UC const *last, T &value,
+       parse_options_t<UC> options) noexcept {
+    return from_chars_float_advanced(first, last, value, options);
+  }
+};
+
+template <> struct from_chars_advanced_caller<2> {
+  template <typename T, typename UC>
+  simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 static from_chars_result_t<UC>
+  call(UC const *first, UC const *last, T &value,
+       parse_options_t<UC> options) noexcept {
+    return from_chars_int_advanced(first, last, value, options);
+  }
+};
+
+template <typename T, typename UC>
+simdjson_fastfloat_really_inline SIMDJSON_FASTFLOAT_CONSTEXPR20 from_chars_result_t<UC>
+from_chars_advanced(UC const *first, UC const *last, T &value,
+                    parse_options_t<UC> options) noexcept {
+  return from_chars_advanced_caller<
+      size_t(is_supported_float_type<T>::value) +
+      2 * size_t(is_supported_integer_type<T>::value)>::call(first, last, value,
+                                                             options);
+}
+
+} // namespace simdjson_fast_float
+
+#endif
+
+/* end file simdjson/internal/fast_float.h */
+#include <array>
+#include <cstdint>
+#include <meta>
+#include <limits>
+#include <string_view>
+
+#include <algorithm>
+#include <array>
+#include <charconv>
+#include <cstdint>
+#include <expected>
+#include <meta>
+#include <string>
+#include <string_view>
+#include <vector>
+
+#define simdjson_consteval_error(...)                                          \
+  {                                                                            \
+    std::abort();                                                              \
+  }
+
+namespace simdjson {
+namespace compile_time {
+
+/**
+ * Namespace for number parsing utilities.
+ * We seek to provide exact compile-time number parsing functions.
+ * Correct rounding of floating-point numbers is not a trivial matter, and it is
+ * harder still in a constant expression, where the runtime parser's memcpy and
+ * __uint128_t tricks are unavailable. We hand that part to fast_float, which is
+ * correctly rounded and usable in a constant expression from C++20 onwards.
+ */
+namespace number_parsing {
+
+// Parses a JSON float starting at src. Returns the value and the number of
+// characters consumed.
+//
+// Eisel-Lemire needs somewhere to fall back to: a mantissa of more than 19
+// significant digits, or an input where the truncated product cannot decide the
+// rounding, has to be finished by a slower exact method. A constant expression
+// cannot call the runtime one in src/from_chars.cpp, so we use fast_float, which
+// is correctly rounded in a constant expression and carries its own fallback.
+consteval std::pair<double, size_t> parse_double(const char *src,
+                                                 const char *end) {
+  double value = 0;
+  auto answer = simdjson_fast_float::from_chars_advanced(
+      src, end, value,
+      simdjson_fast_float::parse_options{
+          simdjson_fast_float::chars_format::json});
+  if (answer.ec == std::errc::invalid_argument) {
+    simdjson_consteval_error("Invalid float value");
+  }
+  // fast_float reports result_out_of_range at either edge of the format. An
+  // underflow to zero is a value like any other; an overflow to infinity is not
+  // representable in JSON and was an error here before, so it stays one.
+  // (Comparing against max/lowest rather than calling std::isinf, which is not
+  // usable in a constant expression.)
+  if (value > (std::numeric_limits<double>::max)() ||
+      value < std::numeric_limits<double>::lowest()) {
     simdjson_consteval_error("Overflow while computing the float value");
   }
-  return {d, size_t(p - srcinit)};
+  return {value, size_t(answer.ptr - src)};
 }
+
 } // namespace number_parsing

+consteval auto make_data_member_options(auto&& name_str) {
+  std::meta::data_member_options options{};
+  options.name = std::forward<decltype(name_str)>(name_str);
+  return options;
+}
+
 // JSON string may contain embedded nulls, and C++26 reflection does not yet
 // support std::string_view as a data member type. As a workaround, we define
 // a custom type that holds a const char* and a size.
@@ -187056,7 +244847,8 @@ using class_type = type_builder<meta_info...>::constructed_type;
 /**
  * @brief Variable template for constructing instances with values
  */
-template <typename T, auto... Vs> constexpr auto construct_from = T{Vs...};
+template <typename T, auto... Vs> constexpr T construct_from = T{Vs...};
+

 // in JSON, there are only a few whitespace characters that are allowed
 // outside of objects, arrays, strings, and numbers.
@@ -187130,8 +244922,6 @@ parse_number(std::string_view json,
   // Note that we consider -0 to be an integer unless it has a decimal point or
   // exponent.
   if (is_float) {
-    // It would be cool to use std::from_chars in a consteval context, but it is
-    // not supported yet for floating point types. :-(
     auto [value, offset] =
         number_parsing::parse_double(json.data(), json.data() + json.size());
     if (offset != scope) {
@@ -187146,7 +244936,7 @@ parse_number(std::string_view json,
         std::from_chars(json.data(), json.data() + json.size(), int_value);
     if (res.ec == std::errc()) {
       out = int_value;
-      if ((res.ptr - json.data()) != scope) {
+      if (static_cast<std::size_t>(res.ptr - json.data()) != scope) {
         simdjson_consteval_error(
             "Internal error: cannot agree on the character range of the float");
       }
@@ -187160,7 +244950,7 @@ parse_number(std::string_view json,
         std::from_chars(json.data(), json.data() + json.size(), uint_value);
     if (res.ec == std::errc()) {
       out = uint_value;
-      if ((res.ptr - json.data()) != scope) {
+      if (static_cast<std::size_t>(res.ptr - json.data()) != scope) {
         simdjson_consteval_error(
             "Internal error: cannot agree on the character range of the float");
       }
@@ -187269,7 +245059,7 @@ parse_string(std::string_view json) {
       // present, we have an error (isolated high surrogate), which we
       // tolerate by substituting the substitution_code_point.
       if (end - cursor < 6 || *cursor != '\\' ||
-          *(cursor + 1) != 'u' > 0xFFFF) {
+          *(cursor + 1) != 'u') {
         code_point = substitution_code_point;
       } else {       // we have \u following the high surrogate
         cursor += 2; // skip \u
@@ -187623,10 +245413,19 @@ parse_json_array_impl(const std::string_view json) {
   std::size_t count = values.size() - 1;
   // We assume all elements have the same type as the first element.
   // However, if the array is heterogeneous, we should use std::variant.
+  auto elem_type = std::meta::type_of(values[1]);
+  // String literals reflected via reflect_constant_string have type const
+  // char[N], but when passed as template auto parameters they decay to
+  // const char*.  Use const char* as the element type so that
+  // construct_from can aggregate-initialize the array.
+  if (std::meta::is_array_type(elem_type) &&
+      std::meta::remove_all_extents(elem_type) == ^^const char) {
+    elem_type = ^^const char *;
+  }
   auto array_type = std::meta::substitute(
       ^^std::array,
       {
-          std::meta::type_of(values[1]), std::meta::reflect_constant(count)});
+          elem_type, std::meta::reflect_constant(count)});

   // Create array instance with values
   values[0] = array_type;
@@ -187693,8 +245492,7 @@ parse_json_object_impl(std::string_view json) {
         simdjson_consteval_error("Expected '}'");
       }
       cursor += object_size;
-      auto dms = std::meta::data_member_spec(std::meta::type_of(parsed),
-                                             {.name = field_name});
+      auto dms = std::meta::data_member_spec(std::meta::type_of(parsed), make_data_member_options(field_name));
       members.push_back(std::meta::reflect_constant(dms));
       values.push_back(parsed);

@@ -187703,8 +245501,7 @@ parse_json_object_impl(std::string_view json) {
     case '[': {
       std::string_view value(cursor, end);
       auto [parsed, array_size] = parse_json_array_impl(value);
-      auto dms = std::meta::data_member_spec(std::meta::type_of(parsed),
-                                             {.name = field_name});
+      auto dms = std::meta::data_member_spec(std::meta::type_of(parsed), make_data_member_options(field_name));
       members.push_back(std::meta::reflect_constant(dms));
       values.push_back(parsed);
       if (*(cursor + array_size - 1) != ']') {
@@ -187725,8 +245522,7 @@ parse_json_object_impl(std::string_view json) {
         }
       }
       auto dms =
-          std::meta::data_member_spec(^^const char *, {
-                                                          .name = field_name});
+          std::meta::data_member_spec(^^const char *, make_data_member_options(field_name));
       members.push_back(std::meta::reflect_constant(dms));
       values.push_back(std::meta::reflect_constant_string(value));
       break;
@@ -187737,8 +245533,7 @@ parse_json_object_impl(std::string_view json) {
       }
       cursor += 4;

-      auto dms = std::meta::data_member_spec(^^bool, {
-                                                         .name = field_name});
+      auto dms = std::meta::data_member_spec(^^bool, make_data_member_options(field_name));
       members.push_back(std::meta::reflect_constant(dms));
       values.push_back(std::meta::reflect_constant(true));
       break;
@@ -187749,8 +245544,7 @@ parse_json_object_impl(std::string_view json) {
       }
       cursor += 5;

-      auto dms = std::meta::data_member_spec(^^bool, {
-                                                         .name = field_name});
+      auto dms = std::meta::data_member_spec(^^bool, make_data_member_options(field_name));
       members.push_back(std::meta::reflect_constant(dms));
       values.push_back(std::meta::reflect_constant(false));
       break;
@@ -187761,9 +245555,7 @@ parse_json_object_impl(std::string_view json) {
       }
       cursor += 4;

-      auto dms = std::meta::data_member_spec(^^std::nullptr_t,
-                                             {
-                                                 .name = field_name});
+      auto dms = std::meta::data_member_spec(^^std::nullptr_t, make_data_member_options(field_name));
       members.push_back(std::meta::reflect_constant(dms));
       values.push_back(std::meta::reflect_constant(nullptr));
       break;
@@ -187787,22 +245579,19 @@ parse_json_object_impl(std::string_view json) {
       if (std::holds_alternative<int64_t>(out)) {
         int64_t int_value = std::get<int64_t>(out);
         auto dms =
-            std::meta::data_member_spec(^^int64_t, {
-                                                       .name = field_name});
+            std::meta::data_member_spec(^^int64_t, make_data_member_options(field_name));
         members.push_back(std::meta::reflect_constant(dms));
         values.push_back(std::meta::reflect_constant(int_value));
       } else if (std::holds_alternative<uint64_t>(out)) {
         uint64_t uint_value = std::get<uint64_t>(out);
         auto dms =
-            std::meta::data_member_spec(^^uint64_t, {
-                                                        .name = field_name});
+            std::meta::data_member_spec(^^uint64_t, make_data_member_options(field_name));
         members.push_back(std::meta::reflect_constant(dms));
         values.push_back(std::meta::reflect_constant(uint_value));
       } else {
         double float_value = std::get<double>(out);
         auto dms =
-            std::meta::data_member_spec(^^double, {
-                                                      .name = field_name});
+            std::meta::data_member_spec(^^double, make_data_member_options(field_name));
         members.push_back(std::meta::reflect_constant(dms));
         values.push_back(std::meta::reflect_constant(float_value));
       }
@@ -187847,16 +245636,11 @@ template <constevalutil::fixed_string json_str> consteval auto parse_json() {
                 "Only JSON objects and arrays are supported at the top level, this "
                 "limitation will be lifted in the future.");*/

-  constexpr auto result = json.front() == '['
-                              ? parse_json_array_impl(json)
-                              : parse_json_object_impl(json);
-  return [: result.first :];
-  /*
-  if(json.front() == '[') {
-      return [:parse_json_array_impl(json).first:];
-  } else if(json.front() == '{') {
-   //   return [:parse_json_object_impl(json).first:];
-  }*/
+  if constexpr (json.front() == '[') {
+    return [: parse_json_array_impl(json).first :];
+  } else {
+    return [: parse_json_object_impl(json).first :];
+  }
 }

 } // namespace compile_time